contextos-agents 1.7.0 → 2.0.0-beta.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (309) hide show
  1. package/.agents/AGENTS.md +1 -1
  2. package/.agents/adapters/aider/export.js +117 -99
  3. package/.agents/adapters/claude/export.js +68 -26
  4. package/.agents/adapters/copilot/export.js +90 -53
  5. package/.agents/adapters/cursor/export.js +80 -101
  6. package/.agents/adapters/drift-detector.js +196 -0
  7. package/.agents/adapters/gemini/export.js +76 -45
  8. package/.agents/adapters/pure-compiler.js +443 -0
  9. package/.agents/adapters/zed/export.js +104 -96
  10. package/.agents/compiled/registry.v2.json +504 -0
  11. package/.agents/compiled/registry.v2.sha256 +1 -0
  12. package/.agents/compiler/manifest-compiler.js +963 -0
  13. package/.agents/compiler/vendor/yaml.LICENSE.txt +13 -0
  14. package/.agents/compiler/vendor/yaml.SBOM.json +6 -0
  15. package/.agents/compiler/vendor/yaml.js +139 -0
  16. package/.agents/core/profiles/init.yaml +25 -0
  17. package/.agents/core/skills/context-manager/skill.yaml +3 -5
  18. package/.agents/core/skills/context-os/SKILL.md +3 -6
  19. package/.agents/core/skills/context-os/skill.yaml +3 -8
  20. package/.agents/core/skills/engineering-workflow/skill.yaml +1 -7
  21. package/.agents/core/skills/gemini-precision/skill.yaml +1 -6
  22. package/.agents/core/skills/gstack-roles/skill.yaml +3 -6
  23. package/.agents/core/skills/ponytail-mindset/skill.yaml +1 -7
  24. package/.agents/core/skills/security/skill.yaml +15 -3
  25. package/.agents/ctx.js +574 -111
  26. package/.agents/customization-dx.js +282 -0
  27. package/.agents/doctor.js +719 -64
  28. package/.agents/filesystem/index.js +71 -0
  29. package/.agents/filesystem/journaled-transaction.js +451 -0
  30. package/.agents/filesystem/lockfile-v2.js +275 -0
  31. package/.agents/filesystem/platform-hardening.js +222 -0
  32. package/.agents/filesystem/project-lock.js +218 -0
  33. package/.agents/filesystem/safe-path.js +256 -0
  34. package/.agents/generated/claude/skills/context-os/SKILL.md +1 -1
  35. package/.agents/generated/gemini/skills/context-os/SKILL.md +2 -2
  36. package/.agents/plugins/contextos/plugin.json +1 -1
  37. package/.agents/plugins.js +259 -60
  38. package/.agents/profiles.js +486 -51
  39. package/.agents/resolver.js +50 -534
  40. package/.agents/schemas/attestation.review.v1.json +111 -0
  41. package/.agents/schemas/attestation.verification.v1.json +85 -0
  42. package/.agents/schemas/lockfile.v2.schema.json +134 -0
  43. package/.agents/schemas/profile.v2.schema.json +114 -0
  44. package/.agents/schemas/runtime.thread.v1.json +192 -0
  45. package/.agents/schemas/skill.manifest.v2.json +177 -0
  46. package/.agents/schemas/verification.spec.v1.json +39 -0
  47. package/.agents/schemas/workspace.graph.schema.json +106 -0
  48. package/.agents/transaction-core/event-store.js +288 -0
  49. package/.agents/transaction-core/idempotency.js +129 -0
  50. package/.agents/transaction-core/ipc-lock.js +311 -0
  51. package/.agents/transaction-core/plugin-supply-chain-bundle.js +436 -0
  52. package/.agents/validate.js +44 -1
  53. package/.agents/watch.js +354 -102
  54. package/.agents/workspace/workspace-graph.js +778 -0
  55. package/README.md +59 -387
  56. package/benchmarks/v2/analysis/statistics.js +140 -0
  57. package/benchmarks/v2/analysis/stats.js +69 -0
  58. package/benchmarks/v2/arms/arm-definitions.js +79 -0
  59. package/benchmarks/v2/dataset.schema.json +34 -0
  60. package/benchmarks/v2/evaluators/index.js +25 -0
  61. package/benchmarks/v2/evaluators/verified-success.js +116 -0
  62. package/benchmarks/v2/harness/runner.js +88 -0
  63. package/benchmarks/v2/pilot-tasks.json +392 -0
  64. package/bin/commands/recover.js +88 -0
  65. package/bin/commands/update.js +80 -17
  66. package/bin/commands.js +62 -25
  67. package/bin/index.js +138 -81
  68. package/bin/lib/lockfile.js +5 -3
  69. package/bin/lib/safe-writer.js +34 -3
  70. package/package.json +85 -72
  71. package/registry.json +2 -2
  72. package/registry.v2.schema.json +86 -0
  73. package/.agents/core/profiles/backend.yaml +0 -47
  74. package/.agents/core/profiles/enterprise.yaml +0 -46
  75. package/.agents/core/profiles/frontend.yaml +0 -46
  76. package/.agents/core/profiles/hackathon.yaml +0 -45
  77. package/.agents/core/profiles/mvp.yaml +0 -44
  78. package/.agents/core/profiles/startup.yaml +0 -48
  79. package/.agents/core/skills/adapters/EXAMPLES.md +0 -19
  80. package/.agents/core/skills/adapters/SKILL.md +0 -105
  81. package/.agents/core/skills/adapters/TROUBLESHOOTING.md +0 -7
  82. package/.agents/core/skills/adapters/VALIDATION.json +0 -12
  83. package/.agents/core/skills/adapters/skill.yaml +0 -16
  84. package/.agents/core/skills/architecture-diagrams/SKILL.md +0 -108
  85. package/.agents/core/skills/architecture-diagrams/VALIDATION.json +0 -12
  86. package/.agents/core/skills/architecture-diagrams/skill.yaml +0 -12
  87. package/.agents/core/skills/brutalist-design/SKILL.md +0 -150
  88. package/.agents/core/skills/brutalist-design/VALIDATION.json +0 -12
  89. package/.agents/core/skills/brutalist-design/skill.yaml +0 -12
  90. package/.agents/core/skills/database/EXAMPLES.md +0 -74
  91. package/.agents/core/skills/database/SKILL.md +0 -101
  92. package/.agents/core/skills/database/TROUBLESHOOTING.md +0 -18
  93. package/.agents/core/skills/database/VALIDATION.json +0 -11
  94. package/.agents/core/skills/database/skill.yaml +0 -31
  95. package/.agents/core/skills/ddd/EXAMPLES.md +0 -42
  96. package/.agents/core/skills/ddd/SKILL.md +0 -247
  97. package/.agents/core/skills/ddd/TROUBLESHOOTING.md +0 -19
  98. package/.agents/core/skills/ddd/VALIDATION.json +0 -12
  99. package/.agents/core/skills/ddd/ddd.md +0 -178
  100. package/.agents/core/skills/ddd/skill.yaml +0 -17
  101. package/.agents/core/skills/decisions/EXAMPLES.md +0 -35
  102. package/.agents/core/skills/decisions/SKILL.md +0 -90
  103. package/.agents/core/skills/decisions/TROUBLESHOOTING.md +0 -13
  104. package/.agents/core/skills/decisions/VALIDATION.json +0 -12
  105. package/.agents/core/skills/decisions/skill.yaml +0 -16
  106. package/.agents/core/skills/docker/EXAMPLES.md +0 -56
  107. package/.agents/core/skills/docker/SKILL.md +0 -63
  108. package/.agents/core/skills/docker/TROUBLESHOOTING.md +0 -18
  109. package/.agents/core/skills/docker/VALIDATION.json +0 -11
  110. package/.agents/core/skills/docker/skill.yaml +0 -29
  111. package/.agents/core/skills/fastapi/EXAMPLES.md +0 -36
  112. package/.agents/core/skills/fastapi/SKILL.md +0 -148
  113. package/.agents/core/skills/fastapi/TROUBLESHOOTING.md +0 -19
  114. package/.agents/core/skills/fastapi/VALIDATION.json +0 -12
  115. package/.agents/core/skills/fastapi/fastapi.md +0 -112
  116. package/.agents/core/skills/fastapi/skill.yaml +0 -17
  117. package/.agents/core/skills/generators/EXAMPLES.md +0 -19
  118. package/.agents/core/skills/generators/SKILL.md +0 -112
  119. package/.agents/core/skills/generators/TROUBLESHOOTING.md +0 -7
  120. package/.agents/core/skills/generators/VALIDATION.json +0 -12
  121. package/.agents/core/skills/generators/skill.yaml +0 -25
  122. package/.agents/core/skills/generators/templates/API.md +0 -77
  123. package/.agents/core/skills/generators/templates/ARCHITECTURE.md +0 -70
  124. package/.agents/core/skills/generators/templates/DATABASE.md +0 -42
  125. package/.agents/core/skills/generators/templates/DECISION.md +0 -46
  126. package/.agents/core/skills/generators/templates/PRD.md +0 -67
  127. package/.agents/core/skills/generators/templates/PROJECT_GRAPH.md +0 -56
  128. package/.agents/core/skills/generators/templates/ROADMAP.md +0 -51
  129. package/.agents/core/skills/generators/templates/TASKS.md +0 -43
  130. package/.agents/core/skills/generators/templates/UI.md +0 -73
  131. package/.agents/core/skills/graphify/EXAMPLES.md +0 -73
  132. package/.agents/core/skills/graphify/SKILL.md +0 -130
  133. package/.agents/core/skills/graphify/VALIDATION.json +0 -12
  134. package/.agents/core/skills/graphify/skill.yaml +0 -18
  135. package/.agents/core/skills/impeccable-design/EXAMPLES.md +0 -26
  136. package/.agents/core/skills/impeccable-design/SKILL.md +0 -201
  137. package/.agents/core/skills/impeccable-design/TROUBLESHOOTING.md +0 -19
  138. package/.agents/core/skills/impeccable-design/VALIDATION.json +0 -12
  139. package/.agents/core/skills/impeccable-design/skill.yaml +0 -20
  140. package/.agents/core/skills/interview-me/SKILL.md +0 -97
  141. package/.agents/core/skills/interview-me/VALIDATION.json +0 -12
  142. package/.agents/core/skills/interview-me/skill.yaml +0 -12
  143. package/.agents/core/skills/microservices/EXAMPLES.md +0 -38
  144. package/.agents/core/skills/microservices/SKILL.md +0 -164
  145. package/.agents/core/skills/microservices/TROUBLESHOOTING.md +0 -19
  146. package/.agents/core/skills/microservices/VALIDATION.json +0 -12
  147. package/.agents/core/skills/microservices/microservices.md +0 -119
  148. package/.agents/core/skills/microservices/skill.yaml +0 -17
  149. package/.agents/core/skills/minimalist-design/SKILL.md +0 -113
  150. package/.agents/core/skills/minimalist-design/VALIDATION.json +0 -12
  151. package/.agents/core/skills/minimalist-design/skill.yaml +0 -12
  152. package/.agents/core/skills/nestjs/EXAMPLES.md +0 -40
  153. package/.agents/core/skills/nestjs/SKILL.md +0 -139
  154. package/.agents/core/skills/nestjs/TROUBLESHOOTING.md +0 -19
  155. package/.agents/core/skills/nestjs/VALIDATION.json +0 -12
  156. package/.agents/core/skills/nestjs/nestjs.md +0 -103
  157. package/.agents/core/skills/nestjs/skill.yaml +0 -17
  158. package/.agents/core/skills/nextjs/EXAMPLES.md +0 -40
  159. package/.agents/core/skills/nextjs/SKILL.md +0 -163
  160. package/.agents/core/skills/nextjs/TROUBLESHOOTING.md +0 -19
  161. package/.agents/core/skills/nextjs/VALIDATION.json +0 -12
  162. package/.agents/core/skills/nextjs/nextjs.md +0 -67
  163. package/.agents/core/skills/nextjs/skill.yaml +0 -17
  164. package/.agents/core/skills/node/EXAMPLES.md +0 -80
  165. package/.agents/core/skills/node/SKILL.md +0 -128
  166. package/.agents/core/skills/node/TROUBLESHOOTING.md +0 -19
  167. package/.agents/core/skills/node/VALIDATION.json +0 -12
  168. package/.agents/core/skills/node/node.md +0 -87
  169. package/.agents/core/skills/node/skill.yaml +0 -17
  170. package/.agents/core/skills/performance/EXAMPLES.md +0 -30
  171. package/.agents/core/skills/performance/SKILL.md +0 -75
  172. package/.agents/core/skills/performance/TROUBLESHOOTING.md +0 -19
  173. package/.agents/core/skills/performance/VALIDATION.json +0 -12
  174. package/.agents/core/skills/performance/performance.md +0 -52
  175. package/.agents/core/skills/performance/skill.yaml +0 -17
  176. package/.agents/core/skills/react/EXAMPLES.md +0 -79
  177. package/.agents/core/skills/react/SKILL.md +0 -132
  178. package/.agents/core/skills/react/TROUBLESHOOTING.md +0 -19
  179. package/.agents/core/skills/react/VALIDATION.json +0 -12
  180. package/.agents/core/skills/react/react.md +0 -93
  181. package/.agents/core/skills/react/skill.yaml +0 -17
  182. package/.agents/core/skills/react-best-practices/SKILL.md +0 -155
  183. package/.agents/core/skills/react-best-practices/VALIDATION.json +0 -12
  184. package/.agents/core/skills/react-best-practices/skill.yaml +0 -14
  185. package/.agents/core/skills/redesign-audit/SKILL.md +0 -117
  186. package/.agents/core/skills/redesign-audit/VALIDATION.json +0 -12
  187. package/.agents/core/skills/redesign-audit/skill.yaml +0 -12
  188. package/.agents/core/skills/soft-design/SKILL.md +0 -108
  189. package/.agents/core/skills/soft-design/VALIDATION.json +0 -12
  190. package/.agents/core/skills/soft-design/skill.yaml +0 -12
  191. package/.agents/core/skills/state-management/EXAMPLES.md +0 -56
  192. package/.agents/core/skills/state-management/SKILL.md +0 -48
  193. package/.agents/core/skills/state-management/TROUBLESHOOTING.md +0 -18
  194. package/.agents/core/skills/state-management/VALIDATION.json +0 -11
  195. package/.agents/core/skills/state-management/skill.yaml +0 -28
  196. package/.agents/core/skills/subagent-orchestrator/SKILL.md +0 -117
  197. package/.agents/core/skills/subagent-orchestrator/VALIDATION.json +0 -12
  198. package/.agents/core/skills/subagent-orchestrator/skill.yaml +0 -12
  199. package/.agents/core/skills/system-design/EXAMPLES.md +0 -75
  200. package/.agents/core/skills/system-design/SKILL.md +0 -419
  201. package/.agents/core/skills/system-design/TROUBLESHOOTING.md +0 -19
  202. package/.agents/core/skills/system-design/VALIDATION.json +0 -12
  203. package/.agents/core/skills/system-design/skill.yaml +0 -20
  204. package/.agents/core/skills/system-design/system-design.md +0 -112
  205. package/.agents/core/skills/testing/EXAMPLES.md +0 -71
  206. package/.agents/core/skills/testing/SKILL.md +0 -70
  207. package/.agents/core/skills/testing/TROUBLESHOOTING.md +0 -18
  208. package/.agents/core/skills/testing/VALIDATION.json +0 -11
  209. package/.agents/core/skills/testing/skill.yaml +0 -32
  210. package/.agents/core/skills/typescript/EXAMPLES.md +0 -64
  211. package/.agents/core/skills/typescript/SKILL.md +0 -112
  212. package/.agents/core/skills/typescript/TROUBLESHOOTING.md +0 -19
  213. package/.agents/core/skills/typescript/VALIDATION.json +0 -12
  214. package/.agents/core/skills/typescript/skill.yaml +0 -17
  215. package/.agents/core/skills/typescript/typescript.md +0 -71
  216. package/.agents/core/skills/ui-design/EXAMPLES.md +0 -21
  217. package/.agents/core/skills/ui-design/SKILL.md +0 -124
  218. package/.agents/core/skills/ui-design/TROUBLESHOOTING.md +0 -19
  219. package/.agents/core/skills/ui-design/VALIDATION.json +0 -12
  220. package/.agents/core/skills/ui-design/skill.yaml +0 -17
  221. package/.agents/core/skills/ui-design/ui.md +0 -88
  222. package/.agents/core/skills/ui-ux-pro/EXAMPLES.md +0 -62
  223. package/.agents/core/skills/ui-ux-pro/SKILL.md +0 -375
  224. package/.agents/core/skills/ui-ux-pro/TROUBLESHOOTING.md +0 -19
  225. package/.agents/core/skills/ui-ux-pro/VALIDATION.json +0 -12
  226. package/.agents/core/skills/ui-ux-pro/skill.yaml +0 -19
  227. package/.agents/core/skills/ux-design/EXAMPLES.md +0 -36
  228. package/.agents/core/skills/ux-design/SKILL.md +0 -116
  229. package/.agents/core/skills/ux-design/TROUBLESHOOTING.md +0 -19
  230. package/.agents/core/skills/ux-design/VALIDATION.json +0 -12
  231. package/.agents/core/skills/ux-design/skill.yaml +0 -17
  232. package/.agents/core/skills/ux-design/ux.md +0 -80
  233. package/.agents/core/skills/vercel-optimize/SKILL.md +0 -83
  234. package/.agents/core/skills/vercel-optimize/VALIDATION.json +0 -12
  235. package/.agents/core/skills/vercel-optimize/scripts/collect-signals.mjs +0 -131
  236. package/.agents/core/skills/vercel-optimize/scripts/gate-investigations.mjs +0 -142
  237. package/.agents/core/skills/vercel-optimize/scripts/merge-signals.mjs +0 -143
  238. package/.agents/core/skills/vercel-optimize/scripts/scan-codebase.mjs +0 -174
  239. package/.agents/core/skills/vercel-optimize/skill.yaml +0 -18
  240. package/.agents/core/skills/web-accessibility/EXAMPLES.md +0 -39
  241. package/.agents/core/skills/web-accessibility/SKILL.md +0 -170
  242. package/.agents/core/skills/web-accessibility/TROUBLESHOOTING.md +0 -19
  243. package/.agents/core/skills/web-accessibility/VALIDATION.json +0 -12
  244. package/.agents/core/skills/web-accessibility/accessibility.md +0 -63
  245. package/.agents/core/skills/web-accessibility/skill.yaml +0 -17
  246. package/.agents/generated/claude/skills/adapters/SKILL.md +0 -126
  247. package/.agents/generated/claude/skills/architecture-diagrams/SKILL.md +0 -101
  248. package/.agents/generated/claude/skills/brutalist-design/SKILL.md +0 -145
  249. package/.agents/generated/claude/skills/database/SKILL.md +0 -191
  250. package/.agents/generated/claude/skills/ddd/SKILL.md +0 -305
  251. package/.agents/generated/claude/skills/decisions/SKILL.md +0 -134
  252. package/.agents/generated/claude/skills/docker/SKILL.md +0 -135
  253. package/.agents/generated/claude/skills/fastapi/SKILL.md +0 -200
  254. package/.agents/generated/claude/skills/generators/SKILL.md +0 -133
  255. package/.agents/generated/claude/skills/graphify/SKILL.md +0 -198
  256. package/.agents/generated/claude/skills/impeccable-design/SKILL.md +0 -241
  257. package/.agents/generated/claude/skills/interview-me/SKILL.md +0 -90
  258. package/.agents/generated/claude/skills/microservices/SKILL.md +0 -218
  259. package/.agents/generated/claude/skills/minimalist-design/SKILL.md +0 -108
  260. package/.agents/generated/claude/skills/nestjs/SKILL.md +0 -195
  261. package/.agents/generated/claude/skills/nextjs/SKILL.md +0 -219
  262. package/.agents/generated/claude/skills/node/SKILL.md +0 -224
  263. package/.agents/generated/claude/skills/performance/SKILL.md +0 -121
  264. package/.agents/generated/claude/skills/react/SKILL.md +0 -227
  265. package/.agents/generated/claude/skills/react-best-practices/SKILL.md +0 -146
  266. package/.agents/generated/claude/skills/redesign-audit/SKILL.md +0 -112
  267. package/.agents/generated/claude/skills/soft-design/SKILL.md +0 -103
  268. package/.agents/generated/claude/skills/state-management/SKILL.md +0 -120
  269. package/.agents/generated/claude/skills/subagent-orchestrator/SKILL.md +0 -110
  270. package/.agents/generated/claude/skills/system-design/SKILL.md +0 -507
  271. package/.agents/generated/claude/skills/testing/SKILL.md +0 -157
  272. package/.agents/generated/claude/skills/typescript/SKILL.md +0 -192
  273. package/.agents/generated/claude/skills/ui-design/SKILL.md +0 -161
  274. package/.agents/generated/claude/skills/ui-ux-pro/SKILL.md +0 -451
  275. package/.agents/generated/claude/skills/ux-design/SKILL.md +0 -168
  276. package/.agents/generated/claude/skills/vercel-optimize/SKILL.md +0 -76
  277. package/.agents/generated/claude/skills/web-accessibility/SKILL.md +0 -225
  278. package/.agents/generated/gemini/skills/adapters/SKILL.md +0 -135
  279. package/.agents/generated/gemini/skills/architecture-diagrams/SKILL.md +0 -107
  280. package/.agents/generated/gemini/skills/brutalist-design/SKILL.md +0 -151
  281. package/.agents/generated/gemini/skills/database/SKILL.md +0 -200
  282. package/.agents/generated/gemini/skills/ddd/SKILL.md +0 -314
  283. package/.agents/generated/gemini/skills/decisions/SKILL.md +0 -143
  284. package/.agents/generated/gemini/skills/docker/SKILL.md +0 -144
  285. package/.agents/generated/gemini/skills/fastapi/SKILL.md +0 -209
  286. package/.agents/generated/gemini/skills/generators/SKILL.md +0 -142
  287. package/.agents/generated/gemini/skills/graphify/SKILL.md +0 -205
  288. package/.agents/generated/gemini/skills/impeccable-design/SKILL.md +0 -250
  289. package/.agents/generated/gemini/skills/interview-me/SKILL.md +0 -96
  290. package/.agents/generated/gemini/skills/microservices/SKILL.md +0 -227
  291. package/.agents/generated/gemini/skills/minimalist-design/SKILL.md +0 -114
  292. package/.agents/generated/gemini/skills/nestjs/SKILL.md +0 -204
  293. package/.agents/generated/gemini/skills/nextjs/SKILL.md +0 -298
  294. package/.agents/generated/gemini/skills/node/SKILL.md +0 -323
  295. package/.agents/generated/gemini/skills/performance/SKILL.md +0 -185
  296. package/.agents/generated/gemini/skills/react/SKILL.md +0 -332
  297. package/.agents/generated/gemini/skills/react-best-practices/SKILL.md +0 -152
  298. package/.agents/generated/gemini/skills/redesign-audit/SKILL.md +0 -118
  299. package/.agents/generated/gemini/skills/soft-design/SKILL.md +0 -109
  300. package/.agents/generated/gemini/skills/state-management/SKILL.md +0 -129
  301. package/.agents/generated/gemini/skills/subagent-orchestrator/SKILL.md +0 -116
  302. package/.agents/generated/gemini/skills/system-design/SKILL.md +0 -631
  303. package/.agents/generated/gemini/skills/testing/SKILL.md +0 -166
  304. package/.agents/generated/gemini/skills/typescript/SKILL.md +0 -275
  305. package/.agents/generated/gemini/skills/ui-design/SKILL.md +0 -170
  306. package/.agents/generated/gemini/skills/ui-ux-pro/SKILL.md +0 -460
  307. package/.agents/generated/gemini/skills/ux-design/SKILL.md +0 -177
  308. package/.agents/generated/gemini/skills/vercel-optimize/SKILL.md +0 -82
  309. package/.agents/generated/gemini/skills/web-accessibility/SKILL.md +0 -300
@@ -0,0 +1,140 @@
1
+ /**
2
+ * benchmarks/v2/analysis/statistics.js
3
+ * ContextOS Benchmark v2 — Statistical Analysis Engine
4
+ *
5
+ * Implements Section 24 of CONTEXTOS_IMPLEMENTATION_PLAN.md:
6
+ * - Primary metric: cost per independently verified successful task
7
+ * - Wilson score 95% confidence intervals for binomial success proportions
8
+ * - Pairwise delta comparison between Arm C/D and Arm B (Concise Checklist comparator)
9
+ * - Structured tabular summary generation
10
+ */
11
+
12
+ 'use strict';
13
+
14
+ /**
15
+ * Calculates Wilson score 95% confidence interval for a proportion.
16
+ *
17
+ * @param {number} successes
18
+ * @param {number} total
19
+ * @param {number} [z=1.96] - 95% confidence z-score
20
+ * @returns {[number, number]} [lower, upper] as percentages [0, 100]
21
+ */
22
+ function calculateWilsonInterval(successes, total, z = 1.96) {
23
+ if (total === 0) return [0, 0];
24
+ const p = successes / total;
25
+ const z2 = z * z;
26
+ const denominator = 1 + z2 / total;
27
+ const center = (p + z2 / (2 * total)) / denominator;
28
+ const margin = (z * Math.sqrt((p * (1 - p)) / total + z2 / (4 * total * total))) / denominator;
29
+
30
+ return [
31
+ Math.max(0, parseFloat(((center - margin) * 100).toFixed(1))),
32
+ Math.min(100, parseFloat(((center + margin) * 100).toFixed(1))),
33
+ ];
34
+ }
35
+
36
+ class BenchmarkStatistics {
37
+ /**
38
+ * Analyzes an array of run outcomes across experimental arms.
39
+ *
40
+ * @param {Array<Object>} runs - List of run records: { armId, success, totalCost, durationMs }
41
+ * @returns {Object} Comprehensive statistical summary
42
+ */
43
+ static analyze(runs) {
44
+ const armsMap = {};
45
+
46
+ for (const run of runs) {
47
+ const { armId, success, totalCost = 0, durationMs = 0 } = run;
48
+ if (!armsMap[armId]) {
49
+ armsMap[armId] = {
50
+ armId,
51
+ totalRuns: 0,
52
+ successfulRuns: 0,
53
+ totalCost: 0,
54
+ totalDurationMs: 0,
55
+ };
56
+ }
57
+
58
+ const item = armsMap[armId];
59
+ item.totalRuns++;
60
+ if (success) item.successfulRuns++;
61
+ item.totalCost += totalCost;
62
+ item.totalDurationMs += durationMs;
63
+ }
64
+
65
+ const armStats = {};
66
+ for (const [armId, d] of Object.entries(armsMap)) {
67
+ const successRate = d.totalRuns > 0 ? (d.successfulRuns / d.totalRuns) * 100 : 0;
68
+ const ci95 = calculateWilsonInterval(d.successfulRuns, d.totalRuns);
69
+ const costPerSuccess = d.successfulRuns > 0 ? d.totalCost / d.successfulRuns : null;
70
+ const avgDurationMs = d.totalRuns > 0 ? d.totalDurationMs / d.totalRuns : 0;
71
+
72
+ armStats[armId] = {
73
+ armId,
74
+ totalRuns: d.totalRuns,
75
+ successfulRuns: d.successfulRuns,
76
+ successRate: parseFloat(successRate.toFixed(1)),
77
+ ci95,
78
+ totalCost: parseFloat(d.totalCost.toFixed(4)),
79
+ costPerVerifiedSuccess: costPerSuccess !== null ? parseFloat(costPerSuccess.toFixed(4)) : null,
80
+ avgDurationMs: Math.round(avgDurationMs),
81
+ };
82
+ }
83
+
84
+ // Pairwise comparisons against Arm B (Concise Checklist)
85
+ const comparator = armStats['arm-b-concise-checklist'];
86
+ const comparisons = {};
87
+
88
+ if (comparator) {
89
+ for (const [armId, stat] of Object.entries(armStats)) {
90
+ if (armId === 'arm-b-concise-checklist') continue;
91
+
92
+ const rateDelta = parseFloat((stat.successRate - comparator.successRate).toFixed(1));
93
+ let costRatio = null;
94
+ if (comparator.costPerVerifiedSuccess && stat.costPerVerifiedSuccess) {
95
+ costRatio = parseFloat((stat.costPerVerifiedSuccess / comparator.costPerVerifiedSuccess).toFixed(2));
96
+ }
97
+
98
+ comparisons[armId] = {
99
+ vsComparator: 'arm-b-concise-checklist',
100
+ successRateDelta: rateDelta,
101
+ costRatio,
102
+ };
103
+ }
104
+ }
105
+
106
+ return {
107
+ timestamp: Date.now(),
108
+ totalRunsAnalyzed: runs.length,
109
+ arms: armStats,
110
+ comparisons,
111
+ };
112
+ }
113
+
114
+ /**
115
+ * Formats statistical analysis into an aligned markdown summary table.
116
+ *
117
+ * @param {Object} analysis
118
+ * @returns {string}
119
+ */
120
+ static formatTable(analysis) {
121
+ const lines = [];
122
+ lines.push('| Arm ID | Runs | Successes | Success Rate (95% CI) | Cost / Verified Success | Avg Latency |');
123
+ lines.push('|---|---:|---:|---:|---:|---:|');
124
+
125
+ for (const arm of Object.values(analysis.arms)) {
126
+ const ciStr = `[${arm.ci95[0]}%, ${arm.ci95[1]}%]`;
127
+ const costStr = arm.costPerVerifiedSuccess !== null ? `$${arm.costPerVerifiedSuccess}` : 'N/A';
128
+ lines.push(
129
+ `| **${arm.armId}** | ${arm.totalRuns} | ${arm.successfulRuns} | ${arm.successRate}% ${ciStr} | ${costStr} | ${arm.avgDurationMs}ms |`
130
+ );
131
+ }
132
+
133
+ return lines.join('\n');
134
+ }
135
+ }
136
+
137
+ module.exports = {
138
+ calculateWilsonInterval,
139
+ BenchmarkStatistics,
140
+ };
@@ -0,0 +1,69 @@
1
+ 'use strict';
2
+
3
+ /**
4
+ * Calculates 95% Confidence Interval for a proportion using the Wald method.
5
+ * @param {number} p - sample proportion (success rate)
6
+ * @param {number} n - sample size
7
+ * @returns {number} Margin of Error
8
+ */
9
+ function calculate95CI(p, n) {
10
+ if (n === 0) return 0;
11
+ // Z-value for 95% confidence is 1.96
12
+ const z = 1.96;
13
+ const standardError = Math.sqrt((p * (1 - p)) / n);
14
+ return z * standardError;
15
+ }
16
+
17
+ /**
18
+ * Analyzes the results of a benchmark run.
19
+ * @param {Array} results Array of execution result objects from the runner
20
+ * @returns {Object} Statistical summary
21
+ */
22
+ function analyzeResults(results) {
23
+ const statsByArm = {};
24
+
25
+ // Group by arm
26
+ for (const res of results) {
27
+ if (!statsByArm[res.armId]) {
28
+ statsByArm[res.armId] = {
29
+ total: 0,
30
+ successes: 0,
31
+ failures: 0,
32
+ totalTokens: 0,
33
+ totalDurationMs: 0
34
+ };
35
+ }
36
+
37
+ const stats = statsByArm[res.armId];
38
+ stats.total++;
39
+ if (res.success) {
40
+ stats.successes++;
41
+ } else {
42
+ stats.failures++;
43
+ }
44
+ stats.totalTokens += res.usage.totalTokens;
45
+ stats.totalDurationMs += res.durationMs;
46
+ }
47
+
48
+ // Calculate rates and CI
49
+ const finalStats = {};
50
+ for (const [armId, stats] of Object.entries(statsByArm)) {
51
+ const successRate = stats.total > 0 ? stats.successes / stats.total : 0;
52
+ const marginOfError = calculate95CI(successRate, stats.total);
53
+
54
+ finalStats[armId] = {
55
+ ...stats,
56
+ successRate: parseFloat((successRate * 100).toFixed(2)),
57
+ confidenceInterval95: `±${(marginOfError * 100).toFixed(2)}%`,
58
+ avgTokensPerTask: stats.total > 0 ? Math.round(stats.totalTokens / stats.total) : 0,
59
+ avgDurationMs: stats.total > 0 ? Math.round(stats.totalDurationMs / stats.total) : 0
60
+ };
61
+ }
62
+
63
+ return finalStats;
64
+ }
65
+
66
+ module.exports = {
67
+ analyzeResults,
68
+ calculate95CI
69
+ };
@@ -0,0 +1,79 @@
1
+ /**
2
+ * benchmarks/v2/arms/arm-definitions.js
3
+ * ContextOS Benchmark v2 — Evaluation Arms Specification
4
+ *
5
+ * Implements Section 24.2 of CONTEXTOS_IMPLEMENTATION_PLAN.md:
6
+ * - Arm A (Vanilla): Neutral baseline system prompt without artificial debuffing
7
+ * - Arm B (Concise Checklist): 10-15 universal engineering rules (~600 tokens) — Primary Comparator
8
+ * - Arm C (ContextOS Core): Dynamic canonical skill resolver without ceremony
9
+ * - Arm D (Full ContextOS): Resolver + risk workflow + isolated runtime verification + review
10
+ */
11
+
12
+ 'use strict';
13
+
14
+ const ARMS = {
15
+ ARM_A_VANILLA: {
16
+ id: 'arm-a-vanilla',
17
+ name: 'Vanilla Baseline',
18
+ description: 'Neutral baseline system prompt without ContextOS rules or checklists.',
19
+ tokenBudgetEstimate: 120,
20
+ buildSystemPrompt: () => {
21
+ return 'You are an expert software engineer. Write clean, complete, working production code that solves the user request.';
22
+ },
23
+ },
24
+
25
+ ARM_B_CONCISE_CHECKLIST: {
26
+ id: 'arm-b-concise-checklist',
27
+ name: 'Concise Checklist',
28
+ description: 'High-density 12-rule engineering checklist (~600 tokens). Primary comparator.',
29
+ tokenBudgetEstimate: 580,
30
+ buildSystemPrompt: () => {
31
+ return [
32
+ 'You are a Senior Staff Engineer.',
33
+ 'Follow this strict engineering checklist:',
34
+ '1. Inspect existing files before editing.',
35
+ '2. Never use placeholders, stubs, or TODO comments.',
36
+ '3. Maintain existing codebase naming conventions and architectural boundaries.',
37
+ '4. Minimize blast radius — modify only files required for the task.',
38
+ '5. Validate all user input and sanitize data paths.',
39
+ '6. Use parameterized queries for database operations.',
40
+ '7. Handle all asynchronous error boundaries explicitly.',
41
+ '8. Write comprehensive unit and integration test assertions.',
42
+ '9. Ensure clean TypeScript typing without any unsafe casts.',
43
+ '10. Verify backward compatibility with existing public APIs.',
44
+ '11. No secrets or credentials in code or commits.',
45
+ '12. Ensure code compiles and all tests pass.',
46
+ ].join('\n');
47
+ },
48
+ },
49
+
50
+ ARM_C_CONTEXTOS_CORE: {
51
+ id: 'arm-c-contextos-core',
52
+ name: 'ContextOS Core (Dynamic Context Selection)',
53
+ description: 'Dynamic canonical resolver selecting exact skills and rules without ceremony.',
54
+ tokenBudgetEstimate: 1400,
55
+ buildSystemPrompt: (resolvedSkills = []) => {
56
+ const skillsHeader = resolvedSkills.length > 0
57
+ ? `[ContextOS Resolved Skills: ${resolvedSkills.join(', ')}]`
58
+ : '[ContextOS Core]';
59
+ return `${skillsHeader}\nExecute task adhering to compiled workspace rules and exact skill invariants.`;
60
+ },
61
+ },
62
+
63
+ ARM_D_FULL_CONTEXTOS: {
64
+ id: 'arm-d-full-contextos',
65
+ name: 'Full ContextOS (Core + Runtime Verification)',
66
+ description: 'Dynamic resolver + risk-based workflow + isolated runtime verification + reviewer pipeline.',
67
+ tokenBudgetEstimate: 2200,
68
+ buildSystemPrompt: (resolvedSkills = [], riskLevel = 'STANDARD') => {
69
+ return [
70
+ `[ContextOS Full Runtime] [RISK: ${riskLevel}] [Skills: ${resolvedSkills.join(', ')}]`,
71
+ 'Execution gated by isolated worktree and mandatory verification attestations before merge readiness.',
72
+ ].join('\n');
73
+ },
74
+ },
75
+ };
76
+
77
+ module.exports = {
78
+ ARMS,
79
+ };
@@ -0,0 +1,34 @@
1
+ {
2
+ "$schema": "http://json-schema.org/draft-07/schema#",
3
+ "title": "ContextOS Benchmark v2 Task Schema",
4
+ "type": "object",
5
+ "properties": {
6
+ "id": {
7
+ "type": "string",
8
+ "description": "Unique immutable identifier for the task"
9
+ },
10
+ "description": {
11
+ "type": "string",
12
+ "description": "The actual prompt provided to the LLM"
13
+ },
14
+ "expectedState": {
15
+ "type": "object",
16
+ "description": "The expected state of the filesystem or output after execution",
17
+ "properties": {
18
+ "filesToExist": {
19
+ "type": "array",
20
+ "items": { "type": "string" }
21
+ },
22
+ "filesToContain": {
23
+ "type": "object",
24
+ "additionalProperties": { "type": "string" }
25
+ }
26
+ }
27
+ },
28
+ "hash": {
29
+ "type": "string",
30
+ "description": "SHA-256 hash of the task for immutability verification"
31
+ }
32
+ },
33
+ "required": ["id", "description", "hash"]
34
+ }
@@ -0,0 +1,25 @@
1
+ 'use strict';
2
+
3
+ /**
4
+ * Evaluates the result of a task execution against the expected state.
5
+ * Since we are using a Mock Provider, the evaluation logic is simplified
6
+ * to just trust the provider's mocked success status. In a real system,
7
+ * this would run ESLint, execute the generated code in a sandbox,
8
+ * and verify the exact AST or output.
9
+ *
10
+ * @param {Object} task The benchmark task definition
11
+ * @param {Object} llmResult The result from the LLM/Mock Provider
12
+ * @returns {Object} { passed: boolean, error: string|null }
13
+ */
14
+ function evaluateTask(task, llmResult) {
15
+ if (llmResult.success) {
16
+ return { passed: true, error: null };
17
+ } else {
18
+ return {
19
+ passed: false,
20
+ error: `Failed to meet expected state for task ${task.id}: ${llmResult.mockedOutput}`
21
+ };
22
+ }
23
+ }
24
+
25
+ module.exports = evaluateTask;
@@ -0,0 +1,116 @@
1
+ /**
2
+ * benchmarks/v2/evaluators/verified-success.js
3
+ * ContextOS Benchmark v2 — Primary Outcome Evaluator
4
+ *
5
+ * Implements Section 24.8 of CONTEXTOS_IMPLEMENTATION_PLAN.md:
6
+ * - Primary metric: independently_verified_success
7
+ * - Evaluates:
8
+ * 1. Patch application / syntax correctness
9
+ * 2. Typecheck / build status
10
+ * 3. Public unit test pass rate
11
+ * 4. Hidden test suite pass rate (isolated oracle)
12
+ * 5. Zero regression on baseline suites
13
+ * 6. Zero P0/P1 security findings or leaked credentials
14
+ * 7. Zero placeholder stubs (TODO, mock placeholders)
15
+ * 8. Budget constraints (tokens, time, turns)
16
+ */
17
+
18
+ 'use strict';
19
+
20
+ const PLACEHOLDER_PATTERNS = [
21
+ /\/\/\s*TODO:\s*implement\b/i,
22
+ /\/\/\s*\.\.\.\s*rest of code\b/i,
23
+ /\bthrow new Error\(["']Not implemented["']\)/i,
24
+ /\bpass\s*#\s*TODO\b/i,
25
+ ];
26
+
27
+ class BenchmarkEvaluator {
28
+ /**
29
+ * Evaluates task run evidence against rigorous quality gates.
30
+ *
31
+ * @param {Object} runEvidence
32
+ * @param {boolean} runEvidence.patchApplied
33
+ * @param {boolean} runEvidence.buildPass
34
+ * @param {number} runEvidence.publicTestsTotal
35
+ * @param {number} runEvidence.publicTestsPassed
36
+ * @param {number} runEvidence.hiddenTestsTotal
37
+ * @param {number} runEvidence.hiddenTestsPassed
38
+ * @param {boolean} [runEvidence.regressions=false]
39
+ * @param {Array<string>} [runEvidence.securityFindings=[]]
40
+ * @param {string} [runEvidence.generatedCode='']
41
+ * @param {Object} [runEvidence.budget]
42
+ * @param {number} [runEvidence.budget.tokensUsed=0]
43
+ * @param {number} [runEvidence.budget.tokenLimit=50000]
44
+ * @param {number} [runEvidence.budget.durationMs=0]
45
+ * @param {number} [runEvidence.budget.timeoutMs=60000]
46
+ * @returns {Object} Evaluation report
47
+ */
48
+ static evaluate(runEvidence) {
49
+ const {
50
+ patchApplied = false,
51
+ buildPass = false,
52
+ publicTestsTotal = 0,
53
+ publicTestsPassed = 0,
54
+ hiddenTestsTotal = 0,
55
+ hiddenTestsPassed = 0,
56
+ regressions = false,
57
+ securityFindings = [],
58
+ generatedCode = '',
59
+ budget = {},
60
+ } = runEvidence;
61
+
62
+ const failures = [];
63
+
64
+ // 1. Patch & build
65
+ if (!patchApplied) failures.push('Patch was not successfully applied');
66
+ if (!buildPass) failures.push('Compilation or typecheck failed');
67
+
68
+ // 2. Tests
69
+ if (publicTestsTotal > 0 && publicTestsPassed < publicTestsTotal) {
70
+ failures.push(`Public tests failed: ${publicTestsPassed}/${publicTestsTotal}`);
71
+ }
72
+ if (hiddenTestsTotal > 0 && hiddenTestsPassed < hiddenTestsTotal) {
73
+ failures.push(`Hidden test oracle failed: ${hiddenTestsPassed}/${hiddenTestsTotal}`);
74
+ }
75
+ if (regressions) {
76
+ failures.push('Regression detected in existing test baseline');
77
+ }
78
+
79
+ // 3. Security
80
+ if (Array.isArray(securityFindings) && securityFindings.length > 0) {
81
+ failures.push(`Security vulnerabilities detected: ${securityFindings.join(', ')}`);
82
+ }
83
+
84
+ // 4. Zero placeholders
85
+ if (generatedCode) {
86
+ for (const pat of PLACEHOLDER_PATTERNS) {
87
+ if (pat.test(generatedCode)) {
88
+ failures.push(`Lazy placeholder detected matching pattern: ${pat.source}`);
89
+ break;
90
+ }
91
+ }
92
+ }
93
+
94
+ // 5. Budget constraints
95
+ if (budget.tokensUsed && budget.tokenLimit && budget.tokensUsed > budget.tokenLimit) {
96
+ failures.push(`Token budget exceeded: ${budget.tokensUsed} > ${budget.tokenLimit}`);
97
+ }
98
+ if (budget.durationMs && budget.timeoutMs && budget.durationMs > budget.timeoutMs) {
99
+ failures.push(`Time budget exceeded: ${budget.durationMs}ms > ${budget.timeoutMs}ms`);
100
+ }
101
+
102
+ const isSuccess = failures.length === 0;
103
+
104
+ return {
105
+ independently_verified_success: isSuccess,
106
+ publicTestRate: publicTestsTotal > 0 ? publicTestsPassed / publicTestsTotal : 1.0,
107
+ hiddenTestRate: hiddenTestsTotal > 0 ? hiddenTestsPassed / hiddenTestsTotal : 1.0,
108
+ failureCount: failures.length,
109
+ failures,
110
+ };
111
+ }
112
+ }
113
+
114
+ module.exports = {
115
+ BenchmarkEvaluator,
116
+ };
@@ -0,0 +1,88 @@
1
+ 'use strict';
2
+
3
+ const crypto = require('crypto');
4
+ const { ARMS } = require('../arms/arm-definitions');
5
+ const evaluateTask = require('../evaluators/index');
6
+
7
+ /**
8
+ * Mock LLM Provider used when real API keys are unavailable.
9
+ * Deterministically simulates success/failure rates based on the arm's capability.
10
+ */
11
+ class MockProvider {
12
+ /**
13
+ * Probability of success for each arm to simulate real-world capability differences.
14
+ */
15
+ static getSuccessProbability(armId) {
16
+ switch (armId) {
17
+ case ARMS.ARM_A_VANILLA.id: return 0.40; // 40% success
18
+ case ARMS.ARM_B_CONCISE_CHECKLIST.id: return 0.65; // 65% success
19
+ case ARMS.ARM_C_CONTEXTOS_CORE.id: return 0.85; // 85% success
20
+ case ARMS.ARM_D_FULL_CONTEXTOS.id: return 0.98; // 98% success
21
+ default: return 0.0;
22
+ }
23
+ }
24
+
25
+ static async execute(task, arm) {
26
+ const probability = this.getSuccessProbability(arm.id);
27
+ // Use hash to deterministically seed pseudo-randomness for the mock run
28
+ const hashInt = parseInt(task.hash.substring(7, 15), 16);
29
+ const successThreshold = probability * 0xffffffff;
30
+
31
+ // Slight artificial delay to simulate API request
32
+ await new Promise(resolve => setTimeout(resolve, 50));
33
+
34
+ const isSuccess = hashInt <= successThreshold;
35
+
36
+ return {
37
+ success: isSuccess,
38
+ mockedOutput: isSuccess
39
+ ? `Successfully generated code for ${task.id}`
40
+ : `Failed or hallucinated output for ${task.id}`,
41
+ usage: {
42
+ promptTokens: arm.tokenBudgetEstimate,
43
+ completionTokens: 300,
44
+ totalTokens: arm.tokenBudgetEstimate + 300
45
+ }
46
+ };
47
+ }
48
+ }
49
+
50
+ /**
51
+ * Executes a single task against a single arm.
52
+ */
53
+ async function runTask(task, armId) {
54
+ const arm = Object.values(ARMS).find(a => a.id === armId);
55
+ if (!arm) throw new Error(`Unknown arm: ${armId}`);
56
+
57
+ const startTime = Date.now();
58
+ const requestId = crypto.randomUUID();
59
+
60
+ // Execute using Mock Provider (replace with real LLM client when API keys are available)
61
+ const llmResult = await MockProvider.execute(task, arm);
62
+
63
+ const durationMs = Date.now() - startTime;
64
+
65
+ // Evaluate the output
66
+ const evaluation = evaluateTask(task, llmResult);
67
+
68
+ return {
69
+ taskId: task.id,
70
+ armId: arm.id,
71
+ requestId,
72
+ timestamp: new Date().toISOString(),
73
+ durationMs,
74
+ success: evaluation.passed,
75
+ error: evaluation.error || null,
76
+ usage: llmResult.usage,
77
+ environmentProvenance: {
78
+ nodeVersion: process.version,
79
+ platform: process.platform,
80
+ engine: 'mock-provider-v1'
81
+ }
82
+ };
83
+ }
84
+
85
+ module.exports = {
86
+ runTask,
87
+ MockProvider
88
+ };