contextos-agents 2.0.0-beta.3 → 2.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.agents/AGENTS.md +53 -33
- package/.agents/adapters/aider/export.js +41 -14
- package/.agents/adapters/claude/export.js +1 -1
- package/.agents/adapters/copilot/export.js +1 -1
- package/.agents/adapters/cursor/export.js +1 -1
- package/.agents/adapters/drift-detector.js +80 -7
- package/.agents/adapters/gemini/export.js +1 -1
- package/.agents/adapters/pure-compiler.js +10 -0
- package/.agents/adapters/shared.js +13 -4
- package/.agents/adapters/zed/export.js +1 -1
- package/.agents/compiled/registry.v2.json +29 -25
- package/.agents/compiled/registry.v2.sha256 +1 -1
- package/.agents/core/skills/context-os/SKILL.md +34 -37
- package/.agents/core/skills/engineering-workflow/SKILL.md +24 -24
- package/.agents/core/skills/gemini-precision/EXAMPLES.md +72 -0
- package/.agents/core/skills/gemini-precision/SKILL.md +2 -1
- package/.agents/core/skills/gemini-precision/TROUBLESHOOTING.md +25 -0
- package/.agents/core/skills/gemini-precision/skill.yaml +2 -0
- package/.agents/core/skills/gstack-roles/SKILL.md +7 -6
- package/.agents/core/skills/security/SKILL.md +44 -16
- package/.agents/core/skills/security/skill.yaml +0 -1
- package/.agents/ctx.js +16 -10
- package/.agents/doctor.js +7 -20
- package/.agents/generated/claude/skills/context-os/SKILL.md +34 -37
- package/.agents/generated/claude/skills/engineering-workflow/SKILL.md +24 -24
- package/.agents/generated/claude/skills/gemini-precision/SKILL.md +102 -1
- package/.agents/generated/claude/skills/gstack-roles/SKILL.md +7 -6
- package/.agents/generated/claude/skills/security/SKILL.md +44 -16
- package/.agents/generated/gemini/skills/context-os/SKILL.md +34 -37
- package/.agents/generated/gemini/skills/engineering-workflow/SKILL.md +24 -24
- package/.agents/generated/gemini/skills/gemini-precision/SKILL.md +105 -1
- package/.agents/generated/gemini/skills/gstack-roles/SKILL.md +7 -6
- package/.agents/generated/gemini/skills/security/SKILL.md +44 -125
- package/.agents/plugins.js +5 -4
- package/.agents/resolver/canonical-resolver.js +7 -7
- package/.agents/validate.js +69 -1
- package/LICENSE +201 -21
- package/NOTICE +4 -0
- package/README.md +87 -34
- package/bin/commands/hook.js +129 -0
- package/bin/commands/scan.js +70 -0
- package/bin/commands/update.js +7 -8
- package/bin/commands.js +39 -1
- package/bin/index.js +144 -33
- package/bin/lib/gate.js +171 -0
- package/bin/lib/git-snapshot.js +187 -0
- package/bin/lib/scan.js +380 -0
- package/package.json +10 -12
- package/.agents/core/skills/security/security.md +0 -106
- package/benchmarks/v2/analysis/statistics.js +0 -140
- package/benchmarks/v2/analysis/stats.js +0 -69
- package/benchmarks/v2/arms/arm-definitions.js +0 -79
- package/benchmarks/v2/dataset.schema.json +0 -34
- package/benchmarks/v2/evaluators/index.js +0 -25
- package/benchmarks/v2/evaluators/verified-success.js +0 -116
- package/benchmarks/v2/harness/runner.js +0 -88
- package/benchmarks/v2/pilot-tasks.json +0 -392
|
@@ -42,7 +42,7 @@ Inspired by [addyosmani/agent-skills](https://github.com/addyosmani/agent-skills
|
|
|
42
42
|
|
|
43
43
|
---
|
|
44
44
|
|
|
45
|
-
### Phase 1: DEFINE
|
|
45
|
+
### Phase 1: DEFINE - /spec
|
|
46
46
|
|
|
47
47
|
**Auto-activates → `[ROLE: Product Manager]`**
|
|
48
48
|
|
|
@@ -76,8 +76,8 @@ Before writing the spec, if there is ambiguity, high blast radius, or multiple a
|
|
|
76
76
|
### Technical Approach
|
|
77
77
|
[Read the relevant code. Understand what changes where.]
|
|
78
78
|
Files affected:
|
|
79
|
-
- `src/X.js`
|
|
80
|
-
- `src/Y.js`
|
|
79
|
+
- `src/X.js` - [what changes]
|
|
80
|
+
- `src/Y.js` - [what changes]
|
|
81
81
|
|
|
82
82
|
### Acceptance Criteria
|
|
83
83
|
- [ ] Given [context], when [action], then [result]
|
|
@@ -90,7 +90,7 @@ Files affected:
|
|
|
90
90
|
|
|
91
91
|
---
|
|
92
92
|
|
|
93
|
-
### Phase 2: PLAN
|
|
93
|
+
### Phase 2: PLAN - /plan
|
|
94
94
|
|
|
95
95
|
**Auto-activates → `[ROLE: Architect]`**
|
|
96
96
|
|
|
@@ -108,7 +108,7 @@ Organize tasks as **Thin Vertical Slices** rather than horizontal layers:
|
|
|
108
108
|
- Each task must be **completable in < 2 hours** of focused work.
|
|
109
109
|
- Each task must be **independently testable**.
|
|
110
110
|
- Tasks must be **ordered by dependency** (blocking tasks first).
|
|
111
|
-
- Each task gets a **test requirement**
|
|
111
|
+
- Each task gets a **test requirement** - no task without a test.
|
|
112
112
|
|
|
113
113
|
#### Plan Template
|
|
114
114
|
|
|
@@ -133,13 +133,13 @@ Organize tasks as **Thin Vertical Slices** rather than horizontal layers:
|
|
|
133
133
|
- [Risk 1]: [Mitigation]
|
|
134
134
|
- [Risk 2]: [Mitigation]
|
|
135
135
|
|
|
136
|
-
### STOP
|
|
136
|
+
### STOP - Awaiting Approval
|
|
137
137
|
Do not proceed to BUILD until this plan is approved.
|
|
138
138
|
```
|
|
139
139
|
|
|
140
140
|
---
|
|
141
141
|
|
|
142
|
-
### Phase 3: BUILD
|
|
142
|
+
### Phase 3: BUILD - /build
|
|
143
143
|
|
|
144
144
|
**Auto-activates → `[ROLE: Senior Developer]`**
|
|
145
145
|
|
|
@@ -147,12 +147,12 @@ Implement one task at a time. Commit after each task.
|
|
|
147
147
|
|
|
148
148
|
#### Build Rules
|
|
149
149
|
|
|
150
|
-
1. **One task per commit**
|
|
151
|
-
2. **Write the test FIRST** (TDD
|
|
152
|
-
3. **No dead code**
|
|
153
|
-
4. **No TODOs in committed code**
|
|
154
|
-
5. **Read before writing**
|
|
155
|
-
6. **Limit the blast radius**
|
|
150
|
+
1. **One task per commit** - atomic, descriptive commit messages.
|
|
151
|
+
2. **Write the test FIRST** (TDD - red-green-refactor).
|
|
152
|
+
3. **No dead code** - if it's not tested, it's not shipped.
|
|
153
|
+
4. **No TODOs in committed code** - resolve or create a tracked issue.
|
|
154
|
+
5. **Read before writing** - understand the surrounding code before changing it.
|
|
155
|
+
6. **Limit the blast radius** - modify ONLY the files explicitly listed in the current task's plan. Do NOT rewrite adjacent components, hooks, or utilities unless strictly required AND approved.
|
|
156
156
|
|
|
157
157
|
#### Commit Message Format
|
|
158
158
|
|
|
@@ -169,7 +169,7 @@ Types: `feat`, `fix`, `refactor`, `test`, `docs`, `chore`
|
|
|
169
169
|
|
|
170
170
|
---
|
|
171
171
|
|
|
172
|
-
### Phase 4: VERIFY
|
|
172
|
+
### Phase 4: VERIFY - /test
|
|
173
173
|
|
|
174
174
|
**Auto-activates → `[ROLE: QA Lead]`**
|
|
175
175
|
|
|
@@ -190,7 +190,7 @@ Tests are proof, not an afterthought.
|
|
|
190
190
|
|
|
191
191
|
For complex React components, prioritize testing _user behavior_ over internal state:
|
|
192
192
|
|
|
193
|
-
- Use **React Testing Library** (`userEvent`, `screen.getByRole`)
|
|
193
|
+
- Use **React Testing Library** (`userEvent`, `screen.getByRole`) - test what the user sees.
|
|
194
194
|
- Use **Playwright** for critical user flows (login, checkout, form submit).
|
|
195
195
|
- Do NOT test implementation details (internal state, private methods, component structure).
|
|
196
196
|
- Focus on: "When user clicks X, does Y appear?" not "Does `useState` hold the right value?"
|
|
@@ -217,7 +217,7 @@ Before moving to Review, verify:
|
|
|
217
217
|
|
|
218
218
|
---
|
|
219
219
|
|
|
220
|
-
### Phase 5: REVIEW
|
|
220
|
+
### Phase 5: REVIEW - /review
|
|
221
221
|
|
|
222
222
|
**Auto-activates → `[ROLE: Staff Engineer]` + `[ROLE: Senior Designer]` for UI tasks**
|
|
223
223
|
|
|
@@ -237,7 +237,7 @@ Inspired by [obra/superpowers](https://github.com/obra/superpowers):
|
|
|
237
237
|
|
|
238
238
|
---
|
|
239
239
|
|
|
240
|
-
### Phase 5.5: SIMPLIFY
|
|
240
|
+
### Phase 5.5: SIMPLIFY - /simplify
|
|
241
241
|
|
|
242
242
|
**Auto-activates → `[ROLE: Staff Engineer]` (Ponytail Mindset)**
|
|
243
243
|
|
|
@@ -250,7 +250,7 @@ Before merging, ruthlessly simplify:
|
|
|
250
250
|
|
|
251
251
|
---
|
|
252
252
|
|
|
253
|
-
### Phase 6: SHIP
|
|
253
|
+
### Phase 6: SHIP - /ship
|
|
254
254
|
|
|
255
255
|
**Auto-activates → `[ROLE: Release Engineer]`**
|
|
256
256
|
|
|
@@ -264,8 +264,7 @@ Only ship when all gates are green.
|
|
|
264
264
|
- [ ] Docs updated (README, API docs, changelogs)
|
|
265
265
|
- [ ] Breaking changes documented
|
|
266
266
|
- [ ] Rollback plan exists
|
|
267
|
-
- [ ] Vercel Preview
|
|
268
|
-
- [ ] Core Web Vitals pass in preview (LCP < 2.5s, CLS < 0.1, INP < 200ms)
|
|
267
|
+
- [ ] Preview / staging deployment verified (if applicable, e.g. Vercel Preview and Core Web Vitals for frontend deployments)
|
|
269
268
|
|
|
270
269
|
#### Operational Self-Improvement
|
|
271
270
|
|
|
@@ -308,6 +307,7 @@ export async function POST(req) {
|
|
|
308
307
|
- [ ] Implementation plan broken down into vertical tasks < 2 hours each.
|
|
309
308
|
- [ ] Tests written before implementation (TDD/BDD).
|
|
310
309
|
- [ ] Code reviewed against correctness, security, performance, and design gates.
|
|
310
|
+
- [ ] Staged security and quality check passes (`contextos scan --staged --enforce`).
|
|
311
311
|
- [ ] Simplification ladder executed before shipping.
|
|
312
312
|
|
|
313
313
|
---
|
|
@@ -334,10 +334,10 @@ export async function POST(req) {
|
|
|
334
334
|
|
|
335
335
|
When completing a task or workflow, you must explicitly report your final status as the last part of your output:
|
|
336
336
|
|
|
337
|
-
- **DONE**
|
|
338
|
-
- **DONE_WITH_CONCERNS**
|
|
339
|
-
- **BLOCKED**
|
|
340
|
-
- **NEEDS_CONTEXT**
|
|
337
|
+
- **DONE** - completed with evidence.
|
|
338
|
+
- **DONE_WITH_CONCERNS** - completed, but list concerns.
|
|
339
|
+
- **BLOCKED** - cannot proceed; state blocker and what was tried.
|
|
340
|
+
- **NEEDS_CONTEXT** - missing info; state exactly what is needed.
|
|
341
341
|
|
|
342
342
|
|
|
343
343
|
<!-- Source: EXAMPLES.md -->
|
|
@@ -48,6 +48,7 @@ Activate whenever:
|
|
|
48
48
|
1. Run the project validator or compiler (`node .agents/ctx.js validate`, `tsc --noEmit`, etc.).
|
|
49
49
|
2. Run unit and integration tests (`npm test`, `pytest`, etc.).
|
|
50
50
|
3. Run linter and formatting checks (`npm run lint:md`, `eslint`, etc.).
|
|
51
|
+
4. Run staged security and quality scanner (`contextos scan --staged --enforce`).
|
|
51
52
|
- If a test or validation fails, do not guess: read the exact error trace, fix the root cause, and re-run until green.
|
|
52
53
|
|
|
53
54
|
### 4. Surgical Blast Radius Containment
|
|
@@ -83,7 +84,7 @@ Activate whenever:
|
|
|
83
84
|
**Eliminate the "black box" by narrating technical decisions.**
|
|
84
85
|
|
|
85
86
|
- Avoid executing long, silent chains of tool calls without user visibility.
|
|
86
|
-
- Provide a concise 1
|
|
87
|
+
- Provide a concise 1-2 sentence transparent status update before key operations:
|
|
87
88
|
- State what was inspected or verified from the code.
|
|
88
89
|
- State the architectural decision made and the immediate next action.
|
|
89
90
|
- Keep narration crisp and actionable without excessive verbosity.
|
|
@@ -169,3 +170,106 @@ export async function updateUser(id, data, session) {
|
|
|
169
170
|
- Enforces the 7-rung ladder of `ponytail-mindset`.
|
|
170
171
|
- Acts as the baseline behavioral guardrail across all Gemini and Antigravity operations.
|
|
171
172
|
|
|
173
|
+
|
|
174
|
+
<!-- Source: EXAMPLES.md -->
|
|
175
|
+
|
|
176
|
+
# gemini-precision Examples - Anti-patterns vs ContextOS Standard
|
|
177
|
+
|
|
178
|
+
## Example 1: Read-Before-Write Invariant (Zero Assumptions)
|
|
179
|
+
|
|
180
|
+
### Anti-pattern: Hallucinated Import and Signature
|
|
181
|
+
|
|
182
|
+
```typescript
|
|
183
|
+
// BAD: Assuming the module exists and export is a default function
|
|
184
|
+
import hashPassword from 'src/utils/crypto';
|
|
185
|
+
const hash = hashPassword(password);
|
|
186
|
+
```
|
|
187
|
+
|
|
188
|
+
### Best practice: ContextOS Standard (Inspected Active Codebase First)
|
|
189
|
+
|
|
190
|
+
```typescript
|
|
191
|
+
// GOOD: Inspected src/lib/auth.ts via view_file before writing code
|
|
192
|
+
import { hashSecret, ARGON2_CONFIG } from '../lib/auth.js';
|
|
193
|
+
const hash = await hashSecret(password, ARGON2_CONFIG);
|
|
194
|
+
```
|
|
195
|
+
|
|
196
|
+
---
|
|
197
|
+
|
|
198
|
+
## Example 2: Zero-Placeholder Invariant (Complete Code Only)
|
|
199
|
+
|
|
200
|
+
### Anti-pattern: Lazy Stubs and Ellipsis Comments
|
|
201
|
+
|
|
202
|
+
```typescript
|
|
203
|
+
// BAD: Emitting incomplete code with TODOs and ellipsis
|
|
204
|
+
export function processTransaction(tx: Transaction) {
|
|
205
|
+
// TODO: validate transaction balance
|
|
206
|
+
// ... rest of implementation stays here ...
|
|
207
|
+
return { status: 'ok' };
|
|
208
|
+
}
|
|
209
|
+
```
|
|
210
|
+
|
|
211
|
+
### Best practice: ContextOS Standard (100% Drop-in Compilable)
|
|
212
|
+
|
|
213
|
+
```typescript
|
|
214
|
+
// GOOD: Fully implemented logic with complete error handling
|
|
215
|
+
export function processTransaction(tx: Transaction): TransactionResult {
|
|
216
|
+
if (!tx.amount || tx.amount <= 0) {
|
|
217
|
+
throw new ValidationError('Transaction amount must be positive');
|
|
218
|
+
}
|
|
219
|
+
if (tx.senderBalance < tx.amount) {
|
|
220
|
+
throw new InsufficientFundsError(tx.senderId, tx.amount);
|
|
221
|
+
}
|
|
222
|
+
return {
|
|
223
|
+
status: 'ok',
|
|
224
|
+
transactionId: tx.id,
|
|
225
|
+
newBalance: tx.senderBalance - tx.amount,
|
|
226
|
+
};
|
|
227
|
+
}
|
|
228
|
+
```
|
|
229
|
+
|
|
230
|
+
---
|
|
231
|
+
|
|
232
|
+
## Example 3: Mandatory Proof-of-Work Invariant
|
|
233
|
+
|
|
234
|
+
### Anti-pattern: Claiming Task Complete Without Evidence
|
|
235
|
+
|
|
236
|
+
```text
|
|
237
|
+
BAD: "I have updated the authentication handler. The code looks correct and is ready to merge."
|
|
238
|
+
```
|
|
239
|
+
|
|
240
|
+
### Best practice: ContextOS Standard (Verified with Automated Gates)
|
|
241
|
+
|
|
242
|
+
```bash
|
|
243
|
+
# GOOD: Run test suite, staged scanner, and consistency checks
|
|
244
|
+
npm test
|
|
245
|
+
contextos scan --staged --enforce
|
|
246
|
+
node .agents/ctx.js validate
|
|
247
|
+
```
|
|
248
|
+
|
|
249
|
+
<!-- Source: TROUBLESHOOTING.md -->
|
|
250
|
+
|
|
251
|
+
# gemini-precision Troubleshooting & Common Failure Modes
|
|
252
|
+
|
|
253
|
+
## 1. Test Failure Investigation (No Guesswork)
|
|
254
|
+
|
|
255
|
+
- **Symptom**: Test fails during `npm test` after code modifications.
|
|
256
|
+
- **Root Cause**: Trying to patch the code without reading the exact assertion diff.
|
|
257
|
+
- **Fix**: Never guess the fix. View the test file line where assertion failed, inspect expected vs actual output, and resolve the root discrepancy.
|
|
258
|
+
|
|
259
|
+
## 2. Accidental Staged Secrets or Placeholders
|
|
260
|
+
|
|
261
|
+
- **Symptom**: `contextos scan --staged --enforce` fails with exit code 1.
|
|
262
|
+
- **Root Cause**: Committed temporary `.env` file or left an unfinished `// TODO: implement later` stub in added lines.
|
|
263
|
+
- **Fix**: Remove or redact the secret before committing. Fully implement the logic or replace the placeholder with an explicit tracked issue rather than committed code stubs.
|
|
264
|
+
|
|
265
|
+
## 3. Scope Creep and Excessive Blast Radius
|
|
266
|
+
|
|
267
|
+
- **Symptom**: Unrelated files reformatted or imports reordered across the repository.
|
|
268
|
+
- **Root Cause**: Full-file rewrite instead of targeted surgical replacement.
|
|
269
|
+
- **Fix**: Use targeted chunks that touch only the lines specified in the task plan. Avoid modifying unrelated styling or formatting.
|
|
270
|
+
|
|
271
|
+
## 4. Forbidden Long Dashes
|
|
272
|
+
|
|
273
|
+
- **Symptom**: Linter or compliance check flags unicode dashes in text.
|
|
274
|
+
- **Root Cause**: Using typography dashes (`\u2014` or `\u2013`) instead of standard ASCII hyphens.
|
|
275
|
+
- **Fix**: Replace all em-dashes and en-dashes with standard ASCII hyphens (` - `) or appropriate punctuation (parentheses, commas, colons).
|
|
@@ -15,7 +15,7 @@ Activate on every task to declare explicit specialist role and mindset before be
|
|
|
15
15
|
|
|
16
16
|
## Rules & Patterns
|
|
17
17
|
|
|
18
|
-
Inspired by [Garry Tan's gstack](https://github.com/garrytan/gstack)
|
|
18
|
+
Inspired by [Garry Tan's gstack](https://github.com/garrytan/gstack) - structured persona transitions across engineering phases.
|
|
19
19
|
|
|
20
20
|
## Core Principle
|
|
21
21
|
|
|
@@ -23,13 +23,14 @@ Inspired by [Garry Tan's gstack](https://github.com/garrytan/gstack) — shippin
|
|
|
23
23
|
|
|
24
24
|
## Role Identification Protocol
|
|
25
25
|
|
|
26
|
-
At the start of each task or major phase switch, declare your role:
|
|
26
|
+
At the start of each task or major phase switch, declare your role using the ContextOS standard format:
|
|
27
27
|
|
|
28
|
-
```
|
|
29
|
-
[
|
|
28
|
+
```text
|
|
29
|
+
[DOMAIN: <Domain>] [PHASE: <Phase>] [ROLE: <Role Name>]
|
|
30
|
+
Skills loaded: <skill-1>, <skill-2>
|
|
30
31
|
```
|
|
31
32
|
|
|
32
|
-
> **Anti-Spam Invariant**: Declare this role **strictly once per phase**. Never prefix intermediate tool calls, file operations, or step updates with role tags.
|
|
33
|
+
> **Anti-Spam Invariant**: Declare this role header **strictly once per phase**. Never prefix intermediate tool calls, file operations, or step updates with role tags.
|
|
33
34
|
|
|
34
35
|
Then execute ONLY within the constraints of that role.
|
|
35
36
|
|
|
@@ -108,7 +109,7 @@ THINK PLAN BUILD REVIEW TEST SHIP
|
|
|
108
109
|
2. **One role at a time.** Don't mix QA and implementation in the same response.
|
|
109
110
|
3. **Declare before acting.** Always state `[ROLE: X]` before switching modes.
|
|
110
111
|
4. **Escalate correctly.** If a QA finds an architectural problem → escalate to Architect role.
|
|
111
|
-
5. **The CEO always goes last on planning**
|
|
112
|
+
5. **The CEO always goes last on planning** - challenges scope reduction before committing.
|
|
112
113
|
|
|
113
114
|
## Example Usage
|
|
114
115
|
|
|
@@ -30,7 +30,7 @@ Activate whenever writing authentication, authorization, session management, dat
|
|
|
30
30
|
|
|
31
31
|
#### 1. Injection (SQL, NoSQL, Command)
|
|
32
32
|
|
|
33
|
-
- Always use parameterized queries
|
|
33
|
+
- Always use parameterized queries - never concatenate user input into SQL or shell commands.
|
|
34
34
|
- Use ORMs (Prisma, Drizzle, SQLAlchemy) with strict schema validation.
|
|
35
35
|
- Validate and sanitize all user input before processing.
|
|
36
36
|
|
|
@@ -79,14 +79,20 @@ When building AI workflows, tools, or MCP servers:
|
|
|
79
79
|
- Never allow untrusted content to override system instructions or tool execution permissions.
|
|
80
80
|
2. **Tool Execution Boundaries**:
|
|
81
81
|
- Destructive operations (database drops, file deletions, payment triggers) MUST require explicit user confirmation.
|
|
82
|
-
- Restrict file system tools to the workspace root
|
|
82
|
+
- Restrict file system tools to the workspace root - block directory traversal (`../`).
|
|
83
83
|
3. **Secret Masking & Output Sanitization**:
|
|
84
84
|
- Scrub API keys (`sk-...`, `Bearer ...`), tokens, and credentials before writing to agent logs or step summaries.
|
|
85
|
+
4. **Sandbox Execution & Write Isolation (Supply-Chain Defense)**:
|
|
86
|
+
- Target code is inspected strictly read-only; never execute target-controlled builds or tests with write access to the repository root.
|
|
87
|
+
- Restrict process write boundaries strictly to an isolated temporary `scratch/` directory.
|
|
88
|
+
- Enforce zero outbound external network access during security audits to prevent secret exfiltration via malicious scripts or dependencies.
|
|
89
|
+
- Promote verified non-secret results to retained `artifacts/` only via trusted parent-side inspection code.
|
|
85
90
|
|
|
86
91
|
---
|
|
87
92
|
|
|
88
93
|
## Code Examples
|
|
89
94
|
|
|
95
|
+
|
|
90
96
|
### Timing-Safe Secret Verification
|
|
91
97
|
|
|
92
98
|
```javascript
|
|
@@ -102,28 +108,50 @@ export function verifyWebhookSignature(payload, signature, secret) {
|
|
|
102
108
|
}
|
|
103
109
|
```
|
|
104
110
|
|
|
105
|
-
###
|
|
111
|
+
### SSRF Prevention Requirements (OWASP Compliant)
|
|
106
112
|
|
|
107
|
-
|
|
108
|
-
|
|
113
|
+
Per [OWASP SSRF Prevention Cheat Sheet](https://cheatsheetseries.owasp.org/cheatsheets/Server_Side_Request_Forgery_Prevention_Cheat_Sheet.html), naive application-level DNS pre-checks followed by standard `fetch(url)` are fundamentally flawed due to DNS rebinding (TOCTOU) and unvalidated HTTP 3xx redirects.
|
|
114
|
+
|
|
115
|
+
#### Mandatory Architectural Controls
|
|
116
|
+
|
|
117
|
+
1. **Network-Layer Defense (Primary)**: For user-supplied arbitrary webhooks or URLs, route all outbound traffic through an isolated egress forward proxy (e.g., Smokescreen, Envoy, Squid) configured with firewall-level IP filters blocking RFC 1918, RFC 6598, link-local (`169.254.169.254`), loopback, and IPv6 local addresses at the socket handshake level.
|
|
118
|
+
2. **Positive Destination Allowlist**: If fetching from known external partners, validate destination hostname against a strict positive allowlist.
|
|
119
|
+
3. **Disable Automatic Redirects**: Always set `redirect: 'error'` or `'manual'`. Never follow HTTP redirects automatically without re-validating the target URL against allowlist rules.
|
|
120
|
+
4. **Protocol & Credential Restrictions**: Enforce `https:` exclusively; reject embedded credentials (`user:pass@host`) and non-standard ports.
|
|
109
121
|
|
|
110
|
-
|
|
122
|
+
```typescript
|
|
123
|
+
/**
|
|
124
|
+
* Verified Allowlist-based HTTP Client (OWASP SSRF Prevention)
|
|
125
|
+
* Enforces HTTPS, strict destination allowlist, and rejects HTTP redirects.
|
|
126
|
+
*/
|
|
127
|
+
export async function fetchFromAllowlist(
|
|
128
|
+
urlString: string,
|
|
129
|
+
allowedHostnames: ReadonlySet<string>,
|
|
130
|
+
options: RequestInit = {}
|
|
131
|
+
): Promise<Response> {
|
|
111
132
|
const parsed = new URL(urlString);
|
|
133
|
+
|
|
134
|
+
// 1. Enforce HTTPS only
|
|
112
135
|
if (parsed.protocol !== 'https:') {
|
|
113
|
-
throw new Error(
|
|
136
|
+
throw new Error(`SSRF blocked: protocol "${parsed.protocol}" is not permitted; HTTPS required`);
|
|
114
137
|
}
|
|
115
138
|
|
|
116
|
-
|
|
117
|
-
if (
|
|
118
|
-
|
|
119
|
-
address.startsWith('10.') ||
|
|
120
|
-
address.startsWith('192.168.') ||
|
|
121
|
-
address === '169.254.169.254'
|
|
122
|
-
) {
|
|
123
|
-
throw new Error('Access to private/metadata IP addresses is blocked');
|
|
139
|
+
// 2. Reject credentials in URL
|
|
140
|
+
if (parsed.username || parsed.password) {
|
|
141
|
+
throw new Error('SSRF blocked: URL credentials (user:password@host) are prohibited');
|
|
124
142
|
}
|
|
125
143
|
|
|
126
|
-
|
|
144
|
+
// 3. Strict positive destination allowlist (prevents internal network probing)
|
|
145
|
+
const normalizedHost = parsed.hostname.toLowerCase();
|
|
146
|
+
if (!allowedHostnames.has(normalizedHost)) {
|
|
147
|
+
throw new Error(`SSRF blocked: destination host "${normalizedHost}" is not in the approved allowlist`);
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
// 4. Disable automatic redirects to prevent redirection to private IPs or metadata endpoints
|
|
151
|
+
return fetch(urlString, {
|
|
152
|
+
...options,
|
|
153
|
+
redirect: 'error'
|
|
154
|
+
});
|
|
127
155
|
}
|
|
128
156
|
```
|
|
129
157
|
|
|
@@ -156,115 +184,6 @@ export async function validateSafeUrl(urlString: string): Promise<URL> {
|
|
|
156
184
|
- Pairs with `system-design` to mandate secure network boundaries and authorization layers.
|
|
157
185
|
|
|
158
186
|
|
|
159
|
-
<!-- Source: security.md -->
|
|
160
|
-
|
|
161
|
-
# Application Security — Best Practices
|
|
162
|
-
|
|
163
|
-
## OWASP Top 10
|
|
164
|
-
|
|
165
|
-
### 1. Injection (SQL, NoSQL, Command)
|
|
166
|
-
|
|
167
|
-
- **Always use parameterized queries** — never concatenate user input into SQL
|
|
168
|
-
- Use ORM (Prisma, SQLAlchemy, TypeORM) — they parameterize by default
|
|
169
|
-
- Validate and sanitize all user input
|
|
170
|
-
|
|
171
|
-
### 2. Broken Authentication
|
|
172
|
-
|
|
173
|
-
- Use bcrypt/argon2 for password hashing (cost factor ≥ 12)
|
|
174
|
-
- JWT: short-lived access tokens (15min), refresh tokens (7 days)
|
|
175
|
-
- Rate limit login attempts
|
|
176
|
-
- Implement account lockout after N failed attempts
|
|
177
|
-
- MFA for sensitive operations
|
|
178
|
-
|
|
179
|
-
### 3. Sensitive Data Exposure
|
|
180
|
-
|
|
181
|
-
- HTTPS everywhere — redirect HTTP to HTTPS
|
|
182
|
-
- Encrypt sensitive data at rest (AES-256)
|
|
183
|
-
- Never log passwords, tokens, or PII
|
|
184
|
-
- Use environment variables for secrets
|
|
185
|
-
|
|
186
|
-
### 4. XML/XXE
|
|
187
|
-
|
|
188
|
-
- Disable external entity processing
|
|
189
|
-
- Use JSON instead of XML where possible
|
|
190
|
-
|
|
191
|
-
### 5. Broken Access Control
|
|
192
|
-
|
|
193
|
-
- Default deny — explicitly grant access
|
|
194
|
-
- RBAC (Role-Based Access Control) or ABAC (Attribute-Based)
|
|
195
|
-
- Check authorization on every request, not just UI
|
|
196
|
-
- Don't rely on client-side validation for security
|
|
197
|
-
|
|
198
|
-
### 6. Security Misconfiguration
|
|
199
|
-
|
|
200
|
-
- Remove default credentials
|
|
201
|
-
- Disable debug mode in production
|
|
202
|
-
- Security headers (see below)
|
|
203
|
-
- Keep dependencies updated
|
|
204
|
-
|
|
205
|
-
### 7. XSS (Cross-Site Scripting)
|
|
206
|
-
|
|
207
|
-
- Escape all output by default
|
|
208
|
-
- Content-Security-Policy header
|
|
209
|
-
- HttpOnly + Secure + SameSite cookies
|
|
210
|
-
- Use framework's built-in XSS protection
|
|
211
|
-
|
|
212
|
-
### 8. Insecure Deserialization
|
|
213
|
-
|
|
214
|
-
- Validate and schema-check all input (Zod, Pydantic, class-validator)
|
|
215
|
-
- Don't deserialize untrusted data
|
|
216
|
-
|
|
217
|
-
### 9. Insufficient Logging
|
|
218
|
-
|
|
219
|
-
- Log all authentication events
|
|
220
|
-
- Log authorization failures
|
|
221
|
-
- Log input validation failures
|
|
222
|
-
- Include request ID for tracing
|
|
223
|
-
|
|
224
|
-
### 10. SSRF (Server-Side Request Forgery)
|
|
225
|
-
|
|
226
|
-
- Validate and allowlist URLs
|
|
227
|
-
- Don't let users control server-side HTTP requests
|
|
228
|
-
|
|
229
|
-
## Security Headers
|
|
230
|
-
|
|
231
|
-
```
|
|
232
|
-
Content-Security-Policy: default-src 'self'
|
|
233
|
-
X-Content-Type-Options: nosniff
|
|
234
|
-
X-Frame-Options: DENY
|
|
235
|
-
Strict-Transport-Security: max-age=31536000; includeSubDomains
|
|
236
|
-
Referrer-Policy: strict-origin-when-cross-origin
|
|
237
|
-
Permissions-Policy: camera=(), microphone=(), geolocation=()
|
|
238
|
-
```
|
|
239
|
-
|
|
240
|
-
## Authentication Patterns
|
|
241
|
-
|
|
242
|
-
### JWT Flow
|
|
243
|
-
|
|
244
|
-
```
|
|
245
|
-
Login → Access Token (15min) + Refresh Token (7d, HttpOnly cookie)
|
|
246
|
-
Request → Authorization: Bearer <access_token>
|
|
247
|
-
Expired → POST /auth/refresh (sends refresh cookie) → new access token
|
|
248
|
-
```
|
|
249
|
-
|
|
250
|
-
### OAuth2 Flow
|
|
251
|
-
|
|
252
|
-
```
|
|
253
|
-
Redirect → Provider (Google, GitHub) → Callback → Create/link user → JWT
|
|
254
|
-
```
|
|
255
|
-
|
|
256
|
-
## Checklist Before Deploy
|
|
257
|
-
|
|
258
|
-
- [ ] All secrets in environment variables
|
|
259
|
-
- [ ] HTTPS enabled
|
|
260
|
-
- [ ] Security headers configured
|
|
261
|
-
- [ ] Input validation on all endpoints
|
|
262
|
-
- [ ] Rate limiting enabled
|
|
263
|
-
- [ ] CORS configured (not `*`)
|
|
264
|
-
- [ ] Error messages don't leak internals
|
|
265
|
-
- [ ] Dependency audit (`npm audit`, `pip audit`)
|
|
266
|
-
- [ ] Logging for security events
|
|
267
|
-
|
|
268
187
|
<!-- Source: EXAMPLES.md -->
|
|
269
188
|
|
|
270
189
|
# Application Security Examples — Anti-patterns vs ContextOS Standard
|
package/.agents/plugins.js
CHANGED
|
@@ -901,17 +901,18 @@ async function search(query) {
|
|
|
901
901
|
* Adapters call this instead of reading CORE_SKILLS directly.
|
|
902
902
|
*/
|
|
903
903
|
function collectAllSkillDirs(targetRoot) {
|
|
904
|
+
const isTargetExplicit = Boolean(targetRoot);
|
|
904
905
|
const root = targetRoot || process.cwd();
|
|
905
906
|
const agentsDir = path.join(root, '.agents');
|
|
906
907
|
const localCore = path.join(agentsDir, 'core', 'skills');
|
|
907
908
|
const localPlugins = path.join(agentsDir, 'plugins');
|
|
908
909
|
|
|
909
|
-
const coreDir = fs.existsSync(localCore) ? localCore : CORE_SKILLS;
|
|
910
|
-
const pluginsDir = fs.existsSync(localPlugins) ? localPlugins : PLUGINS_DIR;
|
|
910
|
+
const coreDir = fs.existsSync(localCore) ? localCore : (isTargetExplicit ? null : CORE_SKILLS);
|
|
911
|
+
const pluginsDir = fs.existsSync(localPlugins) ? localPlugins : (isTargetExplicit ? null : PLUGINS_DIR);
|
|
911
912
|
const dirs = [];
|
|
912
913
|
|
|
913
914
|
// Core skills
|
|
914
|
-
if (fs.existsSync(coreDir)) {
|
|
915
|
+
if (coreDir && fs.existsSync(coreDir)) {
|
|
915
916
|
for (const name of fs.readdirSync(coreDir)) {
|
|
916
917
|
const d = path.join(coreDir, name);
|
|
917
918
|
if (fs.statSync(d).isDirectory()) dirs.push(d);
|
|
@@ -919,7 +920,7 @@ function collectAllSkillDirs(targetRoot) {
|
|
|
919
920
|
}
|
|
920
921
|
|
|
921
922
|
// Plugin skills (supports both standalone skill dirs and plugin bundles with skills/)
|
|
922
|
-
if (fs.existsSync(pluginsDir)) {
|
|
923
|
+
if (pluginsDir && fs.existsSync(pluginsDir)) {
|
|
923
924
|
for (const name of fs.readdirSync(pluginsDir)) {
|
|
924
925
|
const d = path.join(pluginsDir, name);
|
|
925
926
|
if (!fs.statSync(d).isDirectory()) continue;
|
|
@@ -23,15 +23,15 @@ const { WorkspaceGraphBuilder } = require('../workspace/workspace-graph');
|
|
|
23
23
|
// Prompt budget limits (Section 14.4)
|
|
24
24
|
const BUDGET_TIERS = {
|
|
25
25
|
BOOTSTRAP: 1200, // always-on bootstrap budget
|
|
26
|
-
SKILL_SUMMARY:
|
|
27
|
-
SKILL_BODY:
|
|
28
|
-
ROUTINE:
|
|
29
|
-
STANDARD:
|
|
30
|
-
HIGH:
|
|
31
|
-
DESTRUCTIVE:
|
|
26
|
+
SKILL_SUMMARY: 300, // single skill summary limit
|
|
27
|
+
SKILL_BODY: 10000, // single skill body limit (supports rich code examples and rules)
|
|
28
|
+
ROUTINE: 16000, // routine context limit
|
|
29
|
+
STANDARD: 32000, // standard normal compiled context limit (supports 8-10 full skills)
|
|
30
|
+
HIGH: 64000, // high-risk compiled context limit
|
|
31
|
+
DESTRUCTIVE: 128000, // destructive compiled context limit
|
|
32
32
|
};
|
|
33
33
|
|
|
34
|
-
const DEFAULT_CONTEXT_BUDGET_TOKENS =
|
|
34
|
+
const DEFAULT_CONTEXT_BUDGET_TOKENS = 64000;
|
|
35
35
|
|
|
36
36
|
// Risk-based workflows (Section 14.3)
|
|
37
37
|
const WORKFLOW_TEMPLATES = {
|
package/.agents/validate.js
CHANGED
|
@@ -641,10 +641,74 @@ function checkRegistryV2() {
|
|
|
641
641
|
}
|
|
642
642
|
}
|
|
643
643
|
|
|
644
|
+
// ═════════════════════════════════════════════════════════════════════════════
|
|
645
|
+
// CHECK 13 — Catalog Skills Validation (Optional, gated by --catalog)
|
|
646
|
+
// ═════════════════════════════════════════════════════════════════════════════
|
|
647
|
+
function checkCatalogSkills() {
|
|
648
|
+
const catalogSkillsDir = path.join(ROOT, 'catalog', 'skills');
|
|
649
|
+
if (!fs.existsSync(catalogSkillsDir)) {
|
|
650
|
+
error(`[catalog] Catalog skills directory not found: ${catalogSkillsDir}`);
|
|
651
|
+
return;
|
|
652
|
+
}
|
|
653
|
+
|
|
654
|
+
let validated = 0;
|
|
655
|
+
const entries = fs.readdirSync(catalogSkillsDir);
|
|
656
|
+
|
|
657
|
+
for (const name of entries) {
|
|
658
|
+
const dir = path.join(catalogSkillsDir, name);
|
|
659
|
+
if (!fs.statSync(dir).isDirectory()) continue;
|
|
660
|
+
|
|
661
|
+
const skillMd = path.join(dir, 'SKILL.md');
|
|
662
|
+
const yaml = path.join(dir, 'skill.yaml');
|
|
663
|
+
const validationJson = path.join(dir, 'VALIDATION.json');
|
|
664
|
+
|
|
665
|
+
if (!fs.existsSync(skillMd)) {
|
|
666
|
+
error(`[catalog] ${name}: missing SKILL.md`);
|
|
667
|
+
continue;
|
|
668
|
+
}
|
|
669
|
+
|
|
670
|
+
const content = fs.readFileSync(skillMd, 'utf8');
|
|
671
|
+
const fm = parseFrontmatter(content);
|
|
672
|
+
if (!fm || fm.malformed) {
|
|
673
|
+
error(`[catalog] ${name}: malformed YAML frontmatter in SKILL.md`);
|
|
674
|
+
} else {
|
|
675
|
+
if (!yamlField(fm.raw, 'name')) {
|
|
676
|
+
error(`[catalog] ${name}: missing 'name' in SKILL.md frontmatter`);
|
|
677
|
+
}
|
|
678
|
+
if (!yamlField(fm.raw, 'description')) {
|
|
679
|
+
error(`[catalog] ${name}: missing 'description' in SKILL.md frontmatter`);
|
|
680
|
+
}
|
|
681
|
+
}
|
|
682
|
+
|
|
683
|
+
if (!fs.existsSync(yaml)) {
|
|
684
|
+
error(`[catalog] ${name}: missing skill.yaml`);
|
|
685
|
+
} else {
|
|
686
|
+
const yamlText = fs.readFileSync(yaml, 'utf8');
|
|
687
|
+
if (!yamlField(yamlText, 'name') && !yamlField(yamlText, 'id')) {
|
|
688
|
+
error(`[catalog] ${name}: missing 'name' or 'id' in skill.yaml`);
|
|
689
|
+
}
|
|
690
|
+
}
|
|
691
|
+
|
|
692
|
+
if (fs.existsSync(validationJson)) {
|
|
693
|
+
try {
|
|
694
|
+
JSON.parse(fs.readFileSync(validationJson, 'utf8'));
|
|
695
|
+
} catch (err) {
|
|
696
|
+
error(`[catalog] ${name}: invalid JSON in VALIDATION.json: ${err.message}`);
|
|
697
|
+
}
|
|
698
|
+
}
|
|
699
|
+
|
|
700
|
+
validated++;
|
|
701
|
+
}
|
|
702
|
+
|
|
703
|
+
info(`[catalog] ${validated} catalog skills validated (frontmatter, skill.yaml, validation metadata)`);
|
|
704
|
+
}
|
|
705
|
+
|
|
644
706
|
// ═════════════════════════════════════════════════════════════════════════════
|
|
645
707
|
// MAIN
|
|
646
708
|
// ═════════════════════════════════════════════════════════════════════════════
|
|
647
|
-
function run() {
|
|
709
|
+
function run(options = {}) {
|
|
710
|
+
const checkCatalog = Boolean(options.checkCatalog || process.argv.includes('--catalog'));
|
|
711
|
+
|
|
648
712
|
console.log(c.cyan('\nContextOS Validator — scanning skills...\n'));
|
|
649
713
|
|
|
650
714
|
if (!fs.existsSync(CORE_SKILLS)) {
|
|
@@ -670,6 +734,10 @@ function run() {
|
|
|
670
734
|
checkProfilesIntegrity(sourceSkills);
|
|
671
735
|
checkRegistryV2();
|
|
672
736
|
|
|
737
|
+
if (checkCatalog) {
|
|
738
|
+
checkCatalogSkills();
|
|
739
|
+
}
|
|
740
|
+
|
|
673
741
|
const passed = printReport();
|
|
674
742
|
process.exit(passed ? 0 : 1);
|
|
675
743
|
}
|