contextos-agents 1.6.1 → 2.0.0-beta.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (323) hide show
  1. package/.agents/AGENTS.md +6 -1
  2. package/.agents/adapters/aider/export.js +117 -97
  3. package/.agents/adapters/claude/export.js +68 -26
  4. package/.agents/adapters/copilot/export.js +90 -51
  5. package/.agents/adapters/cursor/export.js +83 -68
  6. package/.agents/adapters/drift-detector.js +196 -0
  7. package/.agents/adapters/gemini/export.js +76 -45
  8. package/.agents/adapters/pure-compiler.js +443 -0
  9. package/.agents/adapters/zed/export.js +109 -62
  10. package/.agents/compiled/registry.v2.json +504 -0
  11. package/.agents/compiled/registry.v2.sha256 +1 -0
  12. package/.agents/compiler/manifest-compiler.js +963 -0
  13. package/.agents/compiler/vendor/yaml.LICENSE.txt +13 -0
  14. package/.agents/compiler/vendor/yaml.SBOM.json +6 -0
  15. package/.agents/compiler/vendor/yaml.js +139 -0
  16. package/.agents/core/profiles/init.yaml +25 -0
  17. package/.agents/core/skills/context-manager/references/context-rules.md +59 -0
  18. package/.agents/core/skills/context-manager/skill.yaml +10 -5
  19. package/.agents/core/skills/context-os/SKILL.md +3 -6
  20. package/.agents/core/skills/context-os/skill.yaml +14 -8
  21. package/.agents/core/skills/engineering-workflow/SKILL.md +1 -1
  22. package/.agents/core/skills/engineering-workflow/skill.yaml +7 -7
  23. package/.agents/core/skills/gemini-precision/SKILL.md +4 -0
  24. package/.agents/core/skills/gemini-precision/skill.yaml +5 -6
  25. package/.agents/core/skills/gstack-roles/SKILL.md +3 -1
  26. package/.agents/core/skills/gstack-roles/skill.yaml +9 -6
  27. package/.agents/core/skills/ponytail-mindset/skill.yaml +7 -7
  28. package/.agents/core/skills/security/skill.yaml +21 -2
  29. package/.agents/ctx.js +587 -111
  30. package/.agents/customization-dx.js +282 -0
  31. package/.agents/doctor.js +877 -33
  32. package/.agents/filesystem/index.js +71 -0
  33. package/.agents/filesystem/journaled-transaction.js +451 -0
  34. package/.agents/filesystem/lockfile-v2.js +275 -0
  35. package/.agents/filesystem/platform-hardening.js +222 -0
  36. package/.agents/filesystem/project-lock.js +218 -0
  37. package/.agents/filesystem/safe-path.js +256 -0
  38. package/.agents/generated/claude/skills/context-os/SKILL.md +1 -1
  39. package/.agents/generated/claude/skills/engineering-workflow/SKILL.md +1 -1
  40. package/.agents/generated/claude/skills/gemini-precision/SKILL.md +4 -0
  41. package/.agents/generated/claude/skills/gstack-roles/SKILL.md +3 -1
  42. package/.agents/generated/gemini/skills/context-os/SKILL.md +2 -2
  43. package/.agents/generated/gemini/skills/engineering-workflow/SKILL.md +1 -1
  44. package/.agents/generated/gemini/skills/gemini-precision/SKILL.md +4 -0
  45. package/.agents/generated/gemini/skills/gstack-roles/SKILL.md +3 -1
  46. package/.agents/plugins/contextos/hooks.json +25 -0
  47. package/.agents/plugins/contextos/plugin.json +19 -0
  48. package/.agents/plugins.js +432 -73
  49. package/.agents/profiles.js +507 -46
  50. package/.agents/resolver.js +50 -414
  51. package/.agents/schemas/attestation.review.v1.json +111 -0
  52. package/.agents/schemas/attestation.verification.v1.json +85 -0
  53. package/.agents/schemas/lockfile.v2.schema.json +134 -0
  54. package/.agents/schemas/profile.v2.schema.json +114 -0
  55. package/.agents/schemas/runtime.thread.v1.json +192 -0
  56. package/.agents/schemas/skill.manifest.v2.json +177 -0
  57. package/.agents/schemas/verification.spec.v1.json +39 -0
  58. package/.agents/schemas/workspace.graph.schema.json +106 -0
  59. package/.agents/stats.js +22 -9
  60. package/.agents/transaction-core/event-store.js +288 -0
  61. package/.agents/transaction-core/idempotency.js +129 -0
  62. package/.agents/transaction-core/ipc-lock.js +311 -0
  63. package/.agents/transaction-core/plugin-supply-chain-bundle.js +436 -0
  64. package/.agents/validate.js +143 -14
  65. package/.agents/watch.js +354 -102
  66. package/.agents/workspace/workspace-graph.js +778 -0
  67. package/README.md +59 -387
  68. package/benchmarks/v2/analysis/statistics.js +140 -0
  69. package/benchmarks/v2/analysis/stats.js +69 -0
  70. package/benchmarks/v2/arms/arm-definitions.js +79 -0
  71. package/benchmarks/v2/dataset.schema.json +34 -0
  72. package/benchmarks/v2/evaluators/index.js +25 -0
  73. package/benchmarks/v2/evaluators/verified-success.js +116 -0
  74. package/benchmarks/v2/harness/runner.js +88 -0
  75. package/benchmarks/v2/pilot-tasks.json +392 -0
  76. package/bin/commands/recover.js +88 -0
  77. package/bin/commands/uninstall.js +207 -0
  78. package/bin/commands/update.js +325 -0
  79. package/bin/commands.js +342 -0
  80. package/bin/index.js +326 -149
  81. package/bin/lib/detector.js +106 -0
  82. package/bin/lib/lockfile.js +253 -0
  83. package/bin/lib/safe-writer.js +290 -0
  84. package/package.json +85 -73
  85. package/registry.json +15 -7
  86. package/registry.schema.json +3 -1
  87. package/registry.v2.schema.json +86 -0
  88. package/.agents/core/profiles/backend.yaml +0 -47
  89. package/.agents/core/profiles/enterprise.yaml +0 -46
  90. package/.agents/core/profiles/frontend.yaml +0 -46
  91. package/.agents/core/profiles/hackathon.yaml +0 -45
  92. package/.agents/core/profiles/mvp.yaml +0 -44
  93. package/.agents/core/profiles/startup.yaml +0 -48
  94. package/.agents/core/skills/adapters/EXAMPLES.md +0 -19
  95. package/.agents/core/skills/adapters/SKILL.md +0 -105
  96. package/.agents/core/skills/adapters/TROUBLESHOOTING.md +0 -7
  97. package/.agents/core/skills/adapters/VALIDATION.json +0 -12
  98. package/.agents/core/skills/adapters/skill.yaml +0 -10
  99. package/.agents/core/skills/architecture-diagrams/SKILL.md +0 -108
  100. package/.agents/core/skills/architecture-diagrams/VALIDATION.json +0 -12
  101. package/.agents/core/skills/architecture-diagrams/skill.yaml +0 -8
  102. package/.agents/core/skills/brutalist-design/SKILL.md +0 -150
  103. package/.agents/core/skills/brutalist-design/VALIDATION.json +0 -12
  104. package/.agents/core/skills/brutalist-design/skill.yaml +0 -8
  105. package/.agents/core/skills/database/EXAMPLES.md +0 -74
  106. package/.agents/core/skills/database/SKILL.md +0 -101
  107. package/.agents/core/skills/database/TROUBLESHOOTING.md +0 -18
  108. package/.agents/core/skills/database/VALIDATION.json +0 -11
  109. package/.agents/core/skills/database/skill.yaml +0 -25
  110. package/.agents/core/skills/ddd/EXAMPLES.md +0 -42
  111. package/.agents/core/skills/ddd/SKILL.md +0 -247
  112. package/.agents/core/skills/ddd/TROUBLESHOOTING.md +0 -19
  113. package/.agents/core/skills/ddd/VALIDATION.json +0 -12
  114. package/.agents/core/skills/ddd/ddd.md +0 -178
  115. package/.agents/core/skills/ddd/skill.yaml +0 -10
  116. package/.agents/core/skills/decisions/EXAMPLES.md +0 -35
  117. package/.agents/core/skills/decisions/SKILL.md +0 -90
  118. package/.agents/core/skills/decisions/TROUBLESHOOTING.md +0 -13
  119. package/.agents/core/skills/decisions/VALIDATION.json +0 -12
  120. package/.agents/core/skills/decisions/skill.yaml +0 -10
  121. package/.agents/core/skills/docker/EXAMPLES.md +0 -56
  122. package/.agents/core/skills/docker/SKILL.md +0 -63
  123. package/.agents/core/skills/docker/TROUBLESHOOTING.md +0 -18
  124. package/.agents/core/skills/docker/VALIDATION.json +0 -11
  125. package/.agents/core/skills/docker/skill.yaml +0 -23
  126. package/.agents/core/skills/fastapi/EXAMPLES.md +0 -36
  127. package/.agents/core/skills/fastapi/SKILL.md +0 -148
  128. package/.agents/core/skills/fastapi/TROUBLESHOOTING.md +0 -19
  129. package/.agents/core/skills/fastapi/VALIDATION.json +0 -12
  130. package/.agents/core/skills/fastapi/fastapi.md +0 -112
  131. package/.agents/core/skills/fastapi/skill.yaml +0 -10
  132. package/.agents/core/skills/generators/EXAMPLES.md +0 -19
  133. package/.agents/core/skills/generators/SKILL.md +0 -112
  134. package/.agents/core/skills/generators/TROUBLESHOOTING.md +0 -7
  135. package/.agents/core/skills/generators/VALIDATION.json +0 -12
  136. package/.agents/core/skills/generators/skill.yaml +0 -10
  137. package/.agents/core/skills/generators/templates/API.md +0 -77
  138. package/.agents/core/skills/generators/templates/ARCHITECTURE.md +0 -70
  139. package/.agents/core/skills/generators/templates/DATABASE.md +0 -42
  140. package/.agents/core/skills/generators/templates/DECISION.md +0 -46
  141. package/.agents/core/skills/generators/templates/PRD.md +0 -67
  142. package/.agents/core/skills/generators/templates/PROJECT_GRAPH.md +0 -56
  143. package/.agents/core/skills/generators/templates/ROADMAP.md +0 -51
  144. package/.agents/core/skills/generators/templates/TASKS.md +0 -43
  145. package/.agents/core/skills/generators/templates/UI.md +0 -73
  146. package/.agents/core/skills/graphify/EXAMPLES.md +0 -73
  147. package/.agents/core/skills/graphify/SKILL.md +0 -130
  148. package/.agents/core/skills/graphify/VALIDATION.json +0 -12
  149. package/.agents/core/skills/graphify/skill.yaml +0 -13
  150. package/.agents/core/skills/impeccable-design/EXAMPLES.md +0 -26
  151. package/.agents/core/skills/impeccable-design/SKILL.md +0 -201
  152. package/.agents/core/skills/impeccable-design/TROUBLESHOOTING.md +0 -19
  153. package/.agents/core/skills/impeccable-design/VALIDATION.json +0 -12
  154. package/.agents/core/skills/impeccable-design/skill.yaml +0 -14
  155. package/.agents/core/skills/interview-me/SKILL.md +0 -97
  156. package/.agents/core/skills/interview-me/VALIDATION.json +0 -12
  157. package/.agents/core/skills/interview-me/skill.yaml +0 -8
  158. package/.agents/core/skills/microservices/EXAMPLES.md +0 -38
  159. package/.agents/core/skills/microservices/SKILL.md +0 -164
  160. package/.agents/core/skills/microservices/TROUBLESHOOTING.md +0 -19
  161. package/.agents/core/skills/microservices/VALIDATION.json +0 -12
  162. package/.agents/core/skills/microservices/microservices.md +0 -119
  163. package/.agents/core/skills/microservices/skill.yaml +0 -10
  164. package/.agents/core/skills/minimalist-design/SKILL.md +0 -113
  165. package/.agents/core/skills/minimalist-design/VALIDATION.json +0 -12
  166. package/.agents/core/skills/minimalist-design/skill.yaml +0 -8
  167. package/.agents/core/skills/nestjs/EXAMPLES.md +0 -40
  168. package/.agents/core/skills/nestjs/SKILL.md +0 -139
  169. package/.agents/core/skills/nestjs/TROUBLESHOOTING.md +0 -19
  170. package/.agents/core/skills/nestjs/VALIDATION.json +0 -12
  171. package/.agents/core/skills/nestjs/nestjs.md +0 -103
  172. package/.agents/core/skills/nestjs/skill.yaml +0 -10
  173. package/.agents/core/skills/nextjs/EXAMPLES.md +0 -40
  174. package/.agents/core/skills/nextjs/SKILL.md +0 -163
  175. package/.agents/core/skills/nextjs/TROUBLESHOOTING.md +0 -19
  176. package/.agents/core/skills/nextjs/VALIDATION.json +0 -12
  177. package/.agents/core/skills/nextjs/nextjs.md +0 -67
  178. package/.agents/core/skills/nextjs/skill.yaml +0 -10
  179. package/.agents/core/skills/node/EXAMPLES.md +0 -80
  180. package/.agents/core/skills/node/SKILL.md +0 -128
  181. package/.agents/core/skills/node/TROUBLESHOOTING.md +0 -19
  182. package/.agents/core/skills/node/VALIDATION.json +0 -12
  183. package/.agents/core/skills/node/node.md +0 -87
  184. package/.agents/core/skills/node/skill.yaml +0 -10
  185. package/.agents/core/skills/performance/EXAMPLES.md +0 -30
  186. package/.agents/core/skills/performance/SKILL.md +0 -75
  187. package/.agents/core/skills/performance/TROUBLESHOOTING.md +0 -19
  188. package/.agents/core/skills/performance/VALIDATION.json +0 -12
  189. package/.agents/core/skills/performance/performance.md +0 -52
  190. package/.agents/core/skills/performance/skill.yaml +0 -10
  191. package/.agents/core/skills/react/EXAMPLES.md +0 -79
  192. package/.agents/core/skills/react/SKILL.md +0 -132
  193. package/.agents/core/skills/react/TROUBLESHOOTING.md +0 -19
  194. package/.agents/core/skills/react/VALIDATION.json +0 -12
  195. package/.agents/core/skills/react/react.md +0 -93
  196. package/.agents/core/skills/react/skill.yaml +0 -10
  197. package/.agents/core/skills/react-best-practices/SKILL.md +0 -155
  198. package/.agents/core/skills/react-best-practices/VALIDATION.json +0 -12
  199. package/.agents/core/skills/react-best-practices/skill.yaml +0 -10
  200. package/.agents/core/skills/redesign-audit/SKILL.md +0 -117
  201. package/.agents/core/skills/redesign-audit/VALIDATION.json +0 -12
  202. package/.agents/core/skills/redesign-audit/skill.yaml +0 -8
  203. package/.agents/core/skills/soft-design/SKILL.md +0 -108
  204. package/.agents/core/skills/soft-design/VALIDATION.json +0 -12
  205. package/.agents/core/skills/soft-design/skill.yaml +0 -8
  206. package/.agents/core/skills/state-management/EXAMPLES.md +0 -56
  207. package/.agents/core/skills/state-management/SKILL.md +0 -48
  208. package/.agents/core/skills/state-management/TROUBLESHOOTING.md +0 -18
  209. package/.agents/core/skills/state-management/VALIDATION.json +0 -11
  210. package/.agents/core/skills/state-management/skill.yaml +0 -22
  211. package/.agents/core/skills/subagent-orchestrator/SKILL.md +0 -100
  212. package/.agents/core/skills/subagent-orchestrator/VALIDATION.json +0 -12
  213. package/.agents/core/skills/subagent-orchestrator/skill.yaml +0 -8
  214. package/.agents/core/skills/system-design/EXAMPLES.md +0 -75
  215. package/.agents/core/skills/system-design/SKILL.md +0 -419
  216. package/.agents/core/skills/system-design/TROUBLESHOOTING.md +0 -19
  217. package/.agents/core/skills/system-design/VALIDATION.json +0 -12
  218. package/.agents/core/skills/system-design/skill.yaml +0 -13
  219. package/.agents/core/skills/system-design/system-design.md +0 -112
  220. package/.agents/core/skills/testing/EXAMPLES.md +0 -71
  221. package/.agents/core/skills/testing/SKILL.md +0 -70
  222. package/.agents/core/skills/testing/TROUBLESHOOTING.md +0 -18
  223. package/.agents/core/skills/testing/VALIDATION.json +0 -11
  224. package/.agents/core/skills/testing/skill.yaml +0 -26
  225. package/.agents/core/skills/typescript/EXAMPLES.md +0 -64
  226. package/.agents/core/skills/typescript/SKILL.md +0 -112
  227. package/.agents/core/skills/typescript/TROUBLESHOOTING.md +0 -19
  228. package/.agents/core/skills/typescript/VALIDATION.json +0 -12
  229. package/.agents/core/skills/typescript/skill.yaml +0 -10
  230. package/.agents/core/skills/typescript/typescript.md +0 -71
  231. package/.agents/core/skills/ui-design/EXAMPLES.md +0 -21
  232. package/.agents/core/skills/ui-design/SKILL.md +0 -124
  233. package/.agents/core/skills/ui-design/TROUBLESHOOTING.md +0 -19
  234. package/.agents/core/skills/ui-design/VALIDATION.json +0 -12
  235. package/.agents/core/skills/ui-design/skill.yaml +0 -10
  236. package/.agents/core/skills/ui-design/ui.md +0 -88
  237. package/.agents/core/skills/ui-ux-pro/EXAMPLES.md +0 -62
  238. package/.agents/core/skills/ui-ux-pro/SKILL.md +0 -375
  239. package/.agents/core/skills/ui-ux-pro/TROUBLESHOOTING.md +0 -19
  240. package/.agents/core/skills/ui-ux-pro/VALIDATION.json +0 -12
  241. package/.agents/core/skills/ui-ux-pro/skill.yaml +0 -13
  242. package/.agents/core/skills/ux-design/EXAMPLES.md +0 -36
  243. package/.agents/core/skills/ux-design/SKILL.md +0 -116
  244. package/.agents/core/skills/ux-design/TROUBLESHOOTING.md +0 -19
  245. package/.agents/core/skills/ux-design/VALIDATION.json +0 -12
  246. package/.agents/core/skills/ux-design/skill.yaml +0 -10
  247. package/.agents/core/skills/ux-design/ux.md +0 -80
  248. package/.agents/core/skills/vercel-optimize/SKILL.md +0 -83
  249. package/.agents/core/skills/vercel-optimize/VALIDATION.json +0 -12
  250. package/.agents/core/skills/vercel-optimize/skill.yaml +0 -10
  251. package/.agents/core/skills/web-accessibility/EXAMPLES.md +0 -39
  252. package/.agents/core/skills/web-accessibility/SKILL.md +0 -170
  253. package/.agents/core/skills/web-accessibility/TROUBLESHOOTING.md +0 -19
  254. package/.agents/core/skills/web-accessibility/VALIDATION.json +0 -12
  255. package/.agents/core/skills/web-accessibility/accessibility.md +0 -63
  256. package/.agents/core/skills/web-accessibility/skill.yaml +0 -10
  257. package/.agents/generated/claude/skills/adapters/SKILL.md +0 -126
  258. package/.agents/generated/claude/skills/architecture-diagrams/SKILL.md +0 -101
  259. package/.agents/generated/claude/skills/brutalist-design/SKILL.md +0 -145
  260. package/.agents/generated/claude/skills/database/SKILL.md +0 -191
  261. package/.agents/generated/claude/skills/ddd/SKILL.md +0 -305
  262. package/.agents/generated/claude/skills/decisions/SKILL.md +0 -134
  263. package/.agents/generated/claude/skills/docker/SKILL.md +0 -135
  264. package/.agents/generated/claude/skills/fastapi/SKILL.md +0 -200
  265. package/.agents/generated/claude/skills/generators/SKILL.md +0 -133
  266. package/.agents/generated/claude/skills/graphify/SKILL.md +0 -198
  267. package/.agents/generated/claude/skills/impeccable-design/SKILL.md +0 -241
  268. package/.agents/generated/claude/skills/interview-me/SKILL.md +0 -90
  269. package/.agents/generated/claude/skills/microservices/SKILL.md +0 -218
  270. package/.agents/generated/claude/skills/minimalist-design/SKILL.md +0 -108
  271. package/.agents/generated/claude/skills/nestjs/SKILL.md +0 -195
  272. package/.agents/generated/claude/skills/nextjs/SKILL.md +0 -219
  273. package/.agents/generated/claude/skills/node/SKILL.md +0 -224
  274. package/.agents/generated/claude/skills/performance/SKILL.md +0 -121
  275. package/.agents/generated/claude/skills/react/SKILL.md +0 -227
  276. package/.agents/generated/claude/skills/react-best-practices/SKILL.md +0 -146
  277. package/.agents/generated/claude/skills/redesign-audit/SKILL.md +0 -112
  278. package/.agents/generated/claude/skills/soft-design/SKILL.md +0 -103
  279. package/.agents/generated/claude/skills/state-management/SKILL.md +0 -120
  280. package/.agents/generated/claude/skills/subagent-orchestrator/SKILL.md +0 -93
  281. package/.agents/generated/claude/skills/system-design/SKILL.md +0 -507
  282. package/.agents/generated/claude/skills/testing/SKILL.md +0 -157
  283. package/.agents/generated/claude/skills/typescript/SKILL.md +0 -192
  284. package/.agents/generated/claude/skills/ui-design/SKILL.md +0 -161
  285. package/.agents/generated/claude/skills/ui-ux-pro/SKILL.md +0 -451
  286. package/.agents/generated/claude/skills/ux-design/SKILL.md +0 -168
  287. package/.agents/generated/claude/skills/vercel-optimize/SKILL.md +0 -76
  288. package/.agents/generated/claude/skills/web-accessibility/SKILL.md +0 -225
  289. package/.agents/generated/gemini/skills/adapters/SKILL.md +0 -135
  290. package/.agents/generated/gemini/skills/architecture-diagrams/SKILL.md +0 -107
  291. package/.agents/generated/gemini/skills/brutalist-design/SKILL.md +0 -151
  292. package/.agents/generated/gemini/skills/database/SKILL.md +0 -200
  293. package/.agents/generated/gemini/skills/ddd/SKILL.md +0 -314
  294. package/.agents/generated/gemini/skills/decisions/SKILL.md +0 -143
  295. package/.agents/generated/gemini/skills/docker/SKILL.md +0 -144
  296. package/.agents/generated/gemini/skills/fastapi/SKILL.md +0 -209
  297. package/.agents/generated/gemini/skills/generators/SKILL.md +0 -142
  298. package/.agents/generated/gemini/skills/graphify/SKILL.md +0 -205
  299. package/.agents/generated/gemini/skills/impeccable-design/SKILL.md +0 -250
  300. package/.agents/generated/gemini/skills/interview-me/SKILL.md +0 -96
  301. package/.agents/generated/gemini/skills/microservices/SKILL.md +0 -227
  302. package/.agents/generated/gemini/skills/minimalist-design/SKILL.md +0 -114
  303. package/.agents/generated/gemini/skills/nestjs/SKILL.md +0 -204
  304. package/.agents/generated/gemini/skills/nextjs/SKILL.md +0 -298
  305. package/.agents/generated/gemini/skills/node/SKILL.md +0 -323
  306. package/.agents/generated/gemini/skills/performance/SKILL.md +0 -185
  307. package/.agents/generated/gemini/skills/react/SKILL.md +0 -332
  308. package/.agents/generated/gemini/skills/react-best-practices/SKILL.md +0 -152
  309. package/.agents/generated/gemini/skills/redesign-audit/SKILL.md +0 -118
  310. package/.agents/generated/gemini/skills/soft-design/SKILL.md +0 -109
  311. package/.agents/generated/gemini/skills/state-management/SKILL.md +0 -129
  312. package/.agents/generated/gemini/skills/subagent-orchestrator/SKILL.md +0 -99
  313. package/.agents/generated/gemini/skills/system-design/SKILL.md +0 -631
  314. package/.agents/generated/gemini/skills/testing/SKILL.md +0 -166
  315. package/.agents/generated/gemini/skills/typescript/SKILL.md +0 -275
  316. package/.agents/generated/gemini/skills/ui-design/SKILL.md +0 -170
  317. package/.agents/generated/gemini/skills/ui-ux-pro/SKILL.md +0 -460
  318. package/.agents/generated/gemini/skills/ux-design/SKILL.md +0 -177
  319. package/.agents/generated/gemini/skills/vercel-optimize/SKILL.md +0 -82
  320. package/.agents/generated/gemini/skills/web-accessibility/SKILL.md +0 -300
  321. package/.agents/mcp/runtime.py +0 -470
  322. package/.agents/mcp/server.mjs +0 -189271
  323. package/benchmarks/gemini-issues.js +0 -533
@@ -0,0 +1,140 @@
1
+ /**
2
+ * benchmarks/v2/analysis/statistics.js
3
+ * ContextOS Benchmark v2 — Statistical Analysis Engine
4
+ *
5
+ * Implements Section 24 of CONTEXTOS_IMPLEMENTATION_PLAN.md:
6
+ * - Primary metric: cost per independently verified successful task
7
+ * - Wilson score 95% confidence intervals for binomial success proportions
8
+ * - Pairwise delta comparison between Arm C/D and Arm B (Concise Checklist comparator)
9
+ * - Structured tabular summary generation
10
+ */
11
+
12
+ 'use strict';
13
+
14
+ /**
15
+ * Calculates Wilson score 95% confidence interval for a proportion.
16
+ *
17
+ * @param {number} successes
18
+ * @param {number} total
19
+ * @param {number} [z=1.96] - 95% confidence z-score
20
+ * @returns {[number, number]} [lower, upper] as percentages [0, 100]
21
+ */
22
+ function calculateWilsonInterval(successes, total, z = 1.96) {
23
+ if (total === 0) return [0, 0];
24
+ const p = successes / total;
25
+ const z2 = z * z;
26
+ const denominator = 1 + z2 / total;
27
+ const center = (p + z2 / (2 * total)) / denominator;
28
+ const margin = (z * Math.sqrt((p * (1 - p)) / total + z2 / (4 * total * total))) / denominator;
29
+
30
+ return [
31
+ Math.max(0, parseFloat(((center - margin) * 100).toFixed(1))),
32
+ Math.min(100, parseFloat(((center + margin) * 100).toFixed(1))),
33
+ ];
34
+ }
35
+
36
+ class BenchmarkStatistics {
37
+ /**
38
+ * Analyzes an array of run outcomes across experimental arms.
39
+ *
40
+ * @param {Array<Object>} runs - List of run records: { armId, success, totalCost, durationMs }
41
+ * @returns {Object} Comprehensive statistical summary
42
+ */
43
+ static analyze(runs) {
44
+ const armsMap = {};
45
+
46
+ for (const run of runs) {
47
+ const { armId, success, totalCost = 0, durationMs = 0 } = run;
48
+ if (!armsMap[armId]) {
49
+ armsMap[armId] = {
50
+ armId,
51
+ totalRuns: 0,
52
+ successfulRuns: 0,
53
+ totalCost: 0,
54
+ totalDurationMs: 0,
55
+ };
56
+ }
57
+
58
+ const item = armsMap[armId];
59
+ item.totalRuns++;
60
+ if (success) item.successfulRuns++;
61
+ item.totalCost += totalCost;
62
+ item.totalDurationMs += durationMs;
63
+ }
64
+
65
+ const armStats = {};
66
+ for (const [armId, d] of Object.entries(armsMap)) {
67
+ const successRate = d.totalRuns > 0 ? (d.successfulRuns / d.totalRuns) * 100 : 0;
68
+ const ci95 = calculateWilsonInterval(d.successfulRuns, d.totalRuns);
69
+ const costPerSuccess = d.successfulRuns > 0 ? d.totalCost / d.successfulRuns : null;
70
+ const avgDurationMs = d.totalRuns > 0 ? d.totalDurationMs / d.totalRuns : 0;
71
+
72
+ armStats[armId] = {
73
+ armId,
74
+ totalRuns: d.totalRuns,
75
+ successfulRuns: d.successfulRuns,
76
+ successRate: parseFloat(successRate.toFixed(1)),
77
+ ci95,
78
+ totalCost: parseFloat(d.totalCost.toFixed(4)),
79
+ costPerVerifiedSuccess: costPerSuccess !== null ? parseFloat(costPerSuccess.toFixed(4)) : null,
80
+ avgDurationMs: Math.round(avgDurationMs),
81
+ };
82
+ }
83
+
84
+ // Pairwise comparisons against Arm B (Concise Checklist)
85
+ const comparator = armStats['arm-b-concise-checklist'];
86
+ const comparisons = {};
87
+
88
+ if (comparator) {
89
+ for (const [armId, stat] of Object.entries(armStats)) {
90
+ if (armId === 'arm-b-concise-checklist') continue;
91
+
92
+ const rateDelta = parseFloat((stat.successRate - comparator.successRate).toFixed(1));
93
+ let costRatio = null;
94
+ if (comparator.costPerVerifiedSuccess && stat.costPerVerifiedSuccess) {
95
+ costRatio = parseFloat((stat.costPerVerifiedSuccess / comparator.costPerVerifiedSuccess).toFixed(2));
96
+ }
97
+
98
+ comparisons[armId] = {
99
+ vsComparator: 'arm-b-concise-checklist',
100
+ successRateDelta: rateDelta,
101
+ costRatio,
102
+ };
103
+ }
104
+ }
105
+
106
+ return {
107
+ timestamp: Date.now(),
108
+ totalRunsAnalyzed: runs.length,
109
+ arms: armStats,
110
+ comparisons,
111
+ };
112
+ }
113
+
114
+ /**
115
+ * Formats statistical analysis into an aligned markdown summary table.
116
+ *
117
+ * @param {Object} analysis
118
+ * @returns {string}
119
+ */
120
+ static formatTable(analysis) {
121
+ const lines = [];
122
+ lines.push('| Arm ID | Runs | Successes | Success Rate (95% CI) | Cost / Verified Success | Avg Latency |');
123
+ lines.push('|---|---:|---:|---:|---:|---:|');
124
+
125
+ for (const arm of Object.values(analysis.arms)) {
126
+ const ciStr = `[${arm.ci95[0]}%, ${arm.ci95[1]}%]`;
127
+ const costStr = arm.costPerVerifiedSuccess !== null ? `$${arm.costPerVerifiedSuccess}` : 'N/A';
128
+ lines.push(
129
+ `| **${arm.armId}** | ${arm.totalRuns} | ${arm.successfulRuns} | ${arm.successRate}% ${ciStr} | ${costStr} | ${arm.avgDurationMs}ms |`
130
+ );
131
+ }
132
+
133
+ return lines.join('\n');
134
+ }
135
+ }
136
+
137
+ module.exports = {
138
+ calculateWilsonInterval,
139
+ BenchmarkStatistics,
140
+ };
@@ -0,0 +1,69 @@
1
+ 'use strict';
2
+
3
+ /**
4
+ * Calculates 95% Confidence Interval for a proportion using the Wald method.
5
+ * @param {number} p - sample proportion (success rate)
6
+ * @param {number} n - sample size
7
+ * @returns {number} Margin of Error
8
+ */
9
+ function calculate95CI(p, n) {
10
+ if (n === 0) return 0;
11
+ // Z-value for 95% confidence is 1.96
12
+ const z = 1.96;
13
+ const standardError = Math.sqrt((p * (1 - p)) / n);
14
+ return z * standardError;
15
+ }
16
+
17
+ /**
18
+ * Analyzes the results of a benchmark run.
19
+ * @param {Array} results Array of execution result objects from the runner
20
+ * @returns {Object} Statistical summary
21
+ */
22
+ function analyzeResults(results) {
23
+ const statsByArm = {};
24
+
25
+ // Group by arm
26
+ for (const res of results) {
27
+ if (!statsByArm[res.armId]) {
28
+ statsByArm[res.armId] = {
29
+ total: 0,
30
+ successes: 0,
31
+ failures: 0,
32
+ totalTokens: 0,
33
+ totalDurationMs: 0
34
+ };
35
+ }
36
+
37
+ const stats = statsByArm[res.armId];
38
+ stats.total++;
39
+ if (res.success) {
40
+ stats.successes++;
41
+ } else {
42
+ stats.failures++;
43
+ }
44
+ stats.totalTokens += res.usage.totalTokens;
45
+ stats.totalDurationMs += res.durationMs;
46
+ }
47
+
48
+ // Calculate rates and CI
49
+ const finalStats = {};
50
+ for (const [armId, stats] of Object.entries(statsByArm)) {
51
+ const successRate = stats.total > 0 ? stats.successes / stats.total : 0;
52
+ const marginOfError = calculate95CI(successRate, stats.total);
53
+
54
+ finalStats[armId] = {
55
+ ...stats,
56
+ successRate: parseFloat((successRate * 100).toFixed(2)),
57
+ confidenceInterval95: `±${(marginOfError * 100).toFixed(2)}%`,
58
+ avgTokensPerTask: stats.total > 0 ? Math.round(stats.totalTokens / stats.total) : 0,
59
+ avgDurationMs: stats.total > 0 ? Math.round(stats.totalDurationMs / stats.total) : 0
60
+ };
61
+ }
62
+
63
+ return finalStats;
64
+ }
65
+
66
+ module.exports = {
67
+ analyzeResults,
68
+ calculate95CI
69
+ };
@@ -0,0 +1,79 @@
1
+ /**
2
+ * benchmarks/v2/arms/arm-definitions.js
3
+ * ContextOS Benchmark v2 — Evaluation Arms Specification
4
+ *
5
+ * Implements Section 24.2 of CONTEXTOS_IMPLEMENTATION_PLAN.md:
6
+ * - Arm A (Vanilla): Neutral baseline system prompt without artificial debuffing
7
+ * - Arm B (Concise Checklist): 10-15 universal engineering rules (~600 tokens) — Primary Comparator
8
+ * - Arm C (ContextOS Core): Dynamic canonical skill resolver without ceremony
9
+ * - Arm D (Full ContextOS): Resolver + risk workflow + isolated runtime verification + review
10
+ */
11
+
12
+ 'use strict';
13
+
14
+ const ARMS = {
15
+ ARM_A_VANILLA: {
16
+ id: 'arm-a-vanilla',
17
+ name: 'Vanilla Baseline',
18
+ description: 'Neutral baseline system prompt without ContextOS rules or checklists.',
19
+ tokenBudgetEstimate: 120,
20
+ buildSystemPrompt: () => {
21
+ return 'You are an expert software engineer. Write clean, complete, working production code that solves the user request.';
22
+ },
23
+ },
24
+
25
+ ARM_B_CONCISE_CHECKLIST: {
26
+ id: 'arm-b-concise-checklist',
27
+ name: 'Concise Checklist',
28
+ description: 'High-density 12-rule engineering checklist (~600 tokens). Primary comparator.',
29
+ tokenBudgetEstimate: 580,
30
+ buildSystemPrompt: () => {
31
+ return [
32
+ 'You are a Senior Staff Engineer.',
33
+ 'Follow this strict engineering checklist:',
34
+ '1. Inspect existing files before editing.',
35
+ '2. Never use placeholders, stubs, or TODO comments.',
36
+ '3. Maintain existing codebase naming conventions and architectural boundaries.',
37
+ '4. Minimize blast radius — modify only files required for the task.',
38
+ '5. Validate all user input and sanitize data paths.',
39
+ '6. Use parameterized queries for database operations.',
40
+ '7. Handle all asynchronous error boundaries explicitly.',
41
+ '8. Write comprehensive unit and integration test assertions.',
42
+ '9. Ensure clean TypeScript typing without any unsafe casts.',
43
+ '10. Verify backward compatibility with existing public APIs.',
44
+ '11. No secrets or credentials in code or commits.',
45
+ '12. Ensure code compiles and all tests pass.',
46
+ ].join('\n');
47
+ },
48
+ },
49
+
50
+ ARM_C_CONTEXTOS_CORE: {
51
+ id: 'arm-c-contextos-core',
52
+ name: 'ContextOS Core (Dynamic Context Selection)',
53
+ description: 'Dynamic canonical resolver selecting exact skills and rules without ceremony.',
54
+ tokenBudgetEstimate: 1400,
55
+ buildSystemPrompt: (resolvedSkills = []) => {
56
+ const skillsHeader = resolvedSkills.length > 0
57
+ ? `[ContextOS Resolved Skills: ${resolvedSkills.join(', ')}]`
58
+ : '[ContextOS Core]';
59
+ return `${skillsHeader}\nExecute task adhering to compiled workspace rules and exact skill invariants.`;
60
+ },
61
+ },
62
+
63
+ ARM_D_FULL_CONTEXTOS: {
64
+ id: 'arm-d-full-contextos',
65
+ name: 'Full ContextOS (Core + Runtime Verification)',
66
+ description: 'Dynamic resolver + risk-based workflow + isolated runtime verification + reviewer pipeline.',
67
+ tokenBudgetEstimate: 2200,
68
+ buildSystemPrompt: (resolvedSkills = [], riskLevel = 'STANDARD') => {
69
+ return [
70
+ `[ContextOS Full Runtime] [RISK: ${riskLevel}] [Skills: ${resolvedSkills.join(', ')}]`,
71
+ 'Execution gated by isolated worktree and mandatory verification attestations before merge readiness.',
72
+ ].join('\n');
73
+ },
74
+ },
75
+ };
76
+
77
+ module.exports = {
78
+ ARMS,
79
+ };
@@ -0,0 +1,34 @@
1
+ {
2
+ "$schema": "http://json-schema.org/draft-07/schema#",
3
+ "title": "ContextOS Benchmark v2 Task Schema",
4
+ "type": "object",
5
+ "properties": {
6
+ "id": {
7
+ "type": "string",
8
+ "description": "Unique immutable identifier for the task"
9
+ },
10
+ "description": {
11
+ "type": "string",
12
+ "description": "The actual prompt provided to the LLM"
13
+ },
14
+ "expectedState": {
15
+ "type": "object",
16
+ "description": "The expected state of the filesystem or output after execution",
17
+ "properties": {
18
+ "filesToExist": {
19
+ "type": "array",
20
+ "items": { "type": "string" }
21
+ },
22
+ "filesToContain": {
23
+ "type": "object",
24
+ "additionalProperties": { "type": "string" }
25
+ }
26
+ }
27
+ },
28
+ "hash": {
29
+ "type": "string",
30
+ "description": "SHA-256 hash of the task for immutability verification"
31
+ }
32
+ },
33
+ "required": ["id", "description", "hash"]
34
+ }
@@ -0,0 +1,25 @@
1
+ 'use strict';
2
+
3
+ /**
4
+ * Evaluates the result of a task execution against the expected state.
5
+ * Since we are using a Mock Provider, the evaluation logic is simplified
6
+ * to just trust the provider's mocked success status. In a real system,
7
+ * this would run ESLint, execute the generated code in a sandbox,
8
+ * and verify the exact AST or output.
9
+ *
10
+ * @param {Object} task The benchmark task definition
11
+ * @param {Object} llmResult The result from the LLM/Mock Provider
12
+ * @returns {Object} { passed: boolean, error: string|null }
13
+ */
14
+ function evaluateTask(task, llmResult) {
15
+ if (llmResult.success) {
16
+ return { passed: true, error: null };
17
+ } else {
18
+ return {
19
+ passed: false,
20
+ error: `Failed to meet expected state for task ${task.id}: ${llmResult.mockedOutput}`
21
+ };
22
+ }
23
+ }
24
+
25
+ module.exports = evaluateTask;
@@ -0,0 +1,116 @@
1
+ /**
2
+ * benchmarks/v2/evaluators/verified-success.js
3
+ * ContextOS Benchmark v2 — Primary Outcome Evaluator
4
+ *
5
+ * Implements Section 24.8 of CONTEXTOS_IMPLEMENTATION_PLAN.md:
6
+ * - Primary metric: independently_verified_success
7
+ * - Evaluates:
8
+ * 1. Patch application / syntax correctness
9
+ * 2. Typecheck / build status
10
+ * 3. Public unit test pass rate
11
+ * 4. Hidden test suite pass rate (isolated oracle)
12
+ * 5. Zero regression on baseline suites
13
+ * 6. Zero P0/P1 security findings or leaked credentials
14
+ * 7. Zero placeholder stubs (TODO, mock placeholders)
15
+ * 8. Budget constraints (tokens, time, turns)
16
+ */
17
+
18
+ 'use strict';
19
+
20
+ const PLACEHOLDER_PATTERNS = [
21
+ /\/\/\s*TODO:\s*implement\b/i,
22
+ /\/\/\s*\.\.\.\s*rest of code\b/i,
23
+ /\bthrow new Error\(["']Not implemented["']\)/i,
24
+ /\bpass\s*#\s*TODO\b/i,
25
+ ];
26
+
27
+ class BenchmarkEvaluator {
28
+ /**
29
+ * Evaluates task run evidence against rigorous quality gates.
30
+ *
31
+ * @param {Object} runEvidence
32
+ * @param {boolean} runEvidence.patchApplied
33
+ * @param {boolean} runEvidence.buildPass
34
+ * @param {number} runEvidence.publicTestsTotal
35
+ * @param {number} runEvidence.publicTestsPassed
36
+ * @param {number} runEvidence.hiddenTestsTotal
37
+ * @param {number} runEvidence.hiddenTestsPassed
38
+ * @param {boolean} [runEvidence.regressions=false]
39
+ * @param {Array<string>} [runEvidence.securityFindings=[]]
40
+ * @param {string} [runEvidence.generatedCode='']
41
+ * @param {Object} [runEvidence.budget]
42
+ * @param {number} [runEvidence.budget.tokensUsed=0]
43
+ * @param {number} [runEvidence.budget.tokenLimit=50000]
44
+ * @param {number} [runEvidence.budget.durationMs=0]
45
+ * @param {number} [runEvidence.budget.timeoutMs=60000]
46
+ * @returns {Object} Evaluation report
47
+ */
48
+ static evaluate(runEvidence) {
49
+ const {
50
+ patchApplied = false,
51
+ buildPass = false,
52
+ publicTestsTotal = 0,
53
+ publicTestsPassed = 0,
54
+ hiddenTestsTotal = 0,
55
+ hiddenTestsPassed = 0,
56
+ regressions = false,
57
+ securityFindings = [],
58
+ generatedCode = '',
59
+ budget = {},
60
+ } = runEvidence;
61
+
62
+ const failures = [];
63
+
64
+ // 1. Patch & build
65
+ if (!patchApplied) failures.push('Patch was not successfully applied');
66
+ if (!buildPass) failures.push('Compilation or typecheck failed');
67
+
68
+ // 2. Tests
69
+ if (publicTestsTotal > 0 && publicTestsPassed < publicTestsTotal) {
70
+ failures.push(`Public tests failed: ${publicTestsPassed}/${publicTestsTotal}`);
71
+ }
72
+ if (hiddenTestsTotal > 0 && hiddenTestsPassed < hiddenTestsTotal) {
73
+ failures.push(`Hidden test oracle failed: ${hiddenTestsPassed}/${hiddenTestsTotal}`);
74
+ }
75
+ if (regressions) {
76
+ failures.push('Regression detected in existing test baseline');
77
+ }
78
+
79
+ // 3. Security
80
+ if (Array.isArray(securityFindings) && securityFindings.length > 0) {
81
+ failures.push(`Security vulnerabilities detected: ${securityFindings.join(', ')}`);
82
+ }
83
+
84
+ // 4. Zero placeholders
85
+ if (generatedCode) {
86
+ for (const pat of PLACEHOLDER_PATTERNS) {
87
+ if (pat.test(generatedCode)) {
88
+ failures.push(`Lazy placeholder detected matching pattern: ${pat.source}`);
89
+ break;
90
+ }
91
+ }
92
+ }
93
+
94
+ // 5. Budget constraints
95
+ if (budget.tokensUsed && budget.tokenLimit && budget.tokensUsed > budget.tokenLimit) {
96
+ failures.push(`Token budget exceeded: ${budget.tokensUsed} > ${budget.tokenLimit}`);
97
+ }
98
+ if (budget.durationMs && budget.timeoutMs && budget.durationMs > budget.timeoutMs) {
99
+ failures.push(`Time budget exceeded: ${budget.durationMs}ms > ${budget.timeoutMs}ms`);
100
+ }
101
+
102
+ const isSuccess = failures.length === 0;
103
+
104
+ return {
105
+ independently_verified_success: isSuccess,
106
+ publicTestRate: publicTestsTotal > 0 ? publicTestsPassed / publicTestsTotal : 1.0,
107
+ hiddenTestRate: hiddenTestsTotal > 0 ? hiddenTestsPassed / hiddenTestsTotal : 1.0,
108
+ failureCount: failures.length,
109
+ failures,
110
+ };
111
+ }
112
+ }
113
+
114
+ module.exports = {
115
+ BenchmarkEvaluator,
116
+ };
@@ -0,0 +1,88 @@
1
+ 'use strict';
2
+
3
+ const crypto = require('crypto');
4
+ const { ARMS } = require('../arms/arm-definitions');
5
+ const evaluateTask = require('../evaluators/index');
6
+
7
+ /**
8
+ * Mock LLM Provider used when real API keys are unavailable.
9
+ * Deterministically simulates success/failure rates based on the arm's capability.
10
+ */
11
+ class MockProvider {
12
+ /**
13
+ * Probability of success for each arm to simulate real-world capability differences.
14
+ */
15
+ static getSuccessProbability(armId) {
16
+ switch (armId) {
17
+ case ARMS.ARM_A_VANILLA.id: return 0.40; // 40% success
18
+ case ARMS.ARM_B_CONCISE_CHECKLIST.id: return 0.65; // 65% success
19
+ case ARMS.ARM_C_CONTEXTOS_CORE.id: return 0.85; // 85% success
20
+ case ARMS.ARM_D_FULL_CONTEXTOS.id: return 0.98; // 98% success
21
+ default: return 0.0;
22
+ }
23
+ }
24
+
25
+ static async execute(task, arm) {
26
+ const probability = this.getSuccessProbability(arm.id);
27
+ // Use hash to deterministically seed pseudo-randomness for the mock run
28
+ const hashInt = parseInt(task.hash.substring(7, 15), 16);
29
+ const successThreshold = probability * 0xffffffff;
30
+
31
+ // Slight artificial delay to simulate API request
32
+ await new Promise(resolve => setTimeout(resolve, 50));
33
+
34
+ const isSuccess = hashInt <= successThreshold;
35
+
36
+ return {
37
+ success: isSuccess,
38
+ mockedOutput: isSuccess
39
+ ? `Successfully generated code for ${task.id}`
40
+ : `Failed or hallucinated output for ${task.id}`,
41
+ usage: {
42
+ promptTokens: arm.tokenBudgetEstimate,
43
+ completionTokens: 300,
44
+ totalTokens: arm.tokenBudgetEstimate + 300
45
+ }
46
+ };
47
+ }
48
+ }
49
+
50
+ /**
51
+ * Executes a single task against a single arm.
52
+ */
53
+ async function runTask(task, armId) {
54
+ const arm = Object.values(ARMS).find(a => a.id === armId);
55
+ if (!arm) throw new Error(`Unknown arm: ${armId}`);
56
+
57
+ const startTime = Date.now();
58
+ const requestId = crypto.randomUUID();
59
+
60
+ // Execute using Mock Provider (replace with real LLM client when API keys are available)
61
+ const llmResult = await MockProvider.execute(task, arm);
62
+
63
+ const durationMs = Date.now() - startTime;
64
+
65
+ // Evaluate the output
66
+ const evaluation = evaluateTask(task, llmResult);
67
+
68
+ return {
69
+ taskId: task.id,
70
+ armId: arm.id,
71
+ requestId,
72
+ timestamp: new Date().toISOString(),
73
+ durationMs,
74
+ success: evaluation.passed,
75
+ error: evaluation.error || null,
76
+ usage: llmResult.usage,
77
+ environmentProvenance: {
78
+ nodeVersion: process.version,
79
+ platform: process.platform,
80
+ engine: 'mock-provider-v1'
81
+ }
82
+ };
83
+ }
84
+
85
+ module.exports = {
86
+ runTask,
87
+ MockProvider
88
+ };