contextos-agents 1.7.0 → 2.0.0-beta.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (313) hide show
  1. package/.agents/AGENTS.md +16 -50
  2. package/.agents/adapters/aider/export.js +117 -99
  3. package/.agents/adapters/claude/export.js +68 -26
  4. package/.agents/adapters/copilot/export.js +90 -53
  5. package/.agents/adapters/cursor/export.js +100 -104
  6. package/.agents/adapters/drift-detector.js +196 -0
  7. package/.agents/adapters/gemini/export.js +76 -45
  8. package/.agents/adapters/pure-compiler.js +443 -0
  9. package/.agents/adapters/zed/export.js +104 -96
  10. package/.agents/compiled/registry.v2.json +504 -0
  11. package/.agents/compiled/registry.v2.sha256 +1 -0
  12. package/.agents/compiler/manifest-compiler.js +963 -0
  13. package/.agents/compiler/vendor/yaml.LICENSE.txt +13 -0
  14. package/.agents/compiler/vendor/yaml.SBOM.json +6 -0
  15. package/.agents/compiler/vendor/yaml.js +139 -0
  16. package/.agents/core/profiles/init.yaml +25 -0
  17. package/.agents/core/skills/context-manager/skill.yaml +3 -5
  18. package/.agents/core/skills/context-os/SKILL.md +3 -6
  19. package/.agents/core/skills/context-os/skill.yaml +3 -8
  20. package/.agents/core/skills/engineering-workflow/skill.yaml +1 -7
  21. package/.agents/core/skills/gemini-precision/skill.yaml +1 -6
  22. package/.agents/core/skills/gstack-roles/skill.yaml +3 -6
  23. package/.agents/core/skills/ponytail-mindset/skill.yaml +1 -7
  24. package/.agents/core/skills/security/skill.yaml +15 -3
  25. package/.agents/ctx.js +537 -112
  26. package/.agents/customization-dx.js +282 -0
  27. package/.agents/doctor.js +855 -66
  28. package/.agents/filesystem/index.js +71 -0
  29. package/.agents/filesystem/journaled-transaction.js +451 -0
  30. package/.agents/filesystem/lockfile-v2.js +275 -0
  31. package/.agents/filesystem/platform-hardening.js +222 -0
  32. package/.agents/filesystem/project-lock.js +218 -0
  33. package/.agents/filesystem/safe-path.js +256 -0
  34. package/.agents/generated/claude/skills/context-os/SKILL.md +1 -1
  35. package/.agents/generated/gemini/skills/context-os/SKILL.md +2 -2
  36. package/.agents/plugins/contextos/plugin.json +1 -1
  37. package/.agents/plugins.js +278 -64
  38. package/.agents/profiles.js +486 -51
  39. package/.agents/resolver/canonical-resolver.js +1348 -0
  40. package/.agents/resolver.js +50 -534
  41. package/.agents/rules/rule-catalog.js +525 -0
  42. package/.agents/schemas/attestation.review.v1.json +111 -0
  43. package/.agents/schemas/attestation.verification.v1.json +85 -0
  44. package/.agents/schemas/lockfile.v2.schema.json +134 -0
  45. package/.agents/schemas/profile.v2.schema.json +114 -0
  46. package/.agents/schemas/runtime.thread.v1.json +192 -0
  47. package/.agents/schemas/skill.manifest.v2.json +177 -0
  48. package/.agents/schemas/verification.spec.v1.json +39 -0
  49. package/.agents/schemas/workspace.graph.schema.json +106 -0
  50. package/.agents/skills-index.json +6 -166
  51. package/.agents/stats.js +9 -9
  52. package/.agents/transaction-core/event-store.js +288 -0
  53. package/.agents/transaction-core/idempotency.js +129 -0
  54. package/.agents/transaction-core/ipc-lock.js +311 -0
  55. package/.agents/transaction-core/plugin-supply-chain-bundle.js +436 -0
  56. package/.agents/validate.js +44 -1
  57. package/.agents/watch.js +354 -102
  58. package/.agents/workspace/workspace-graph.js +778 -0
  59. package/README.md +59 -387
  60. package/benchmarks/v2/analysis/statistics.js +140 -0
  61. package/benchmarks/v2/analysis/stats.js +69 -0
  62. package/benchmarks/v2/arms/arm-definitions.js +79 -0
  63. package/benchmarks/v2/dataset.schema.json +34 -0
  64. package/benchmarks/v2/evaluators/index.js +25 -0
  65. package/benchmarks/v2/evaluators/verified-success.js +116 -0
  66. package/benchmarks/v2/harness/runner.js +88 -0
  67. package/benchmarks/v2/pilot-tasks.json +392 -0
  68. package/bin/commands/recover.js +88 -0
  69. package/bin/commands/update.js +80 -17
  70. package/bin/commands.js +62 -25
  71. package/bin/index.js +138 -81
  72. package/bin/lib/lockfile.js +5 -3
  73. package/bin/lib/safe-writer.js +34 -3
  74. package/package.json +87 -72
  75. package/registry.json +2 -2
  76. package/registry.v2.schema.json +86 -0
  77. package/.agents/core/profiles/backend.yaml +0 -47
  78. package/.agents/core/profiles/enterprise.yaml +0 -46
  79. package/.agents/core/profiles/frontend.yaml +0 -46
  80. package/.agents/core/profiles/hackathon.yaml +0 -45
  81. package/.agents/core/profiles/mvp.yaml +0 -44
  82. package/.agents/core/profiles/startup.yaml +0 -48
  83. package/.agents/core/skills/adapters/EXAMPLES.md +0 -19
  84. package/.agents/core/skills/adapters/SKILL.md +0 -105
  85. package/.agents/core/skills/adapters/TROUBLESHOOTING.md +0 -7
  86. package/.agents/core/skills/adapters/VALIDATION.json +0 -12
  87. package/.agents/core/skills/adapters/skill.yaml +0 -16
  88. package/.agents/core/skills/architecture-diagrams/SKILL.md +0 -108
  89. package/.agents/core/skills/architecture-diagrams/VALIDATION.json +0 -12
  90. package/.agents/core/skills/architecture-diagrams/skill.yaml +0 -12
  91. package/.agents/core/skills/brutalist-design/SKILL.md +0 -150
  92. package/.agents/core/skills/brutalist-design/VALIDATION.json +0 -12
  93. package/.agents/core/skills/brutalist-design/skill.yaml +0 -12
  94. package/.agents/core/skills/database/EXAMPLES.md +0 -74
  95. package/.agents/core/skills/database/SKILL.md +0 -101
  96. package/.agents/core/skills/database/TROUBLESHOOTING.md +0 -18
  97. package/.agents/core/skills/database/VALIDATION.json +0 -11
  98. package/.agents/core/skills/database/skill.yaml +0 -31
  99. package/.agents/core/skills/ddd/EXAMPLES.md +0 -42
  100. package/.agents/core/skills/ddd/SKILL.md +0 -247
  101. package/.agents/core/skills/ddd/TROUBLESHOOTING.md +0 -19
  102. package/.agents/core/skills/ddd/VALIDATION.json +0 -12
  103. package/.agents/core/skills/ddd/ddd.md +0 -178
  104. package/.agents/core/skills/ddd/skill.yaml +0 -17
  105. package/.agents/core/skills/decisions/EXAMPLES.md +0 -35
  106. package/.agents/core/skills/decisions/SKILL.md +0 -90
  107. package/.agents/core/skills/decisions/TROUBLESHOOTING.md +0 -13
  108. package/.agents/core/skills/decisions/VALIDATION.json +0 -12
  109. package/.agents/core/skills/decisions/skill.yaml +0 -16
  110. package/.agents/core/skills/docker/EXAMPLES.md +0 -56
  111. package/.agents/core/skills/docker/SKILL.md +0 -63
  112. package/.agents/core/skills/docker/TROUBLESHOOTING.md +0 -18
  113. package/.agents/core/skills/docker/VALIDATION.json +0 -11
  114. package/.agents/core/skills/docker/skill.yaml +0 -29
  115. package/.agents/core/skills/fastapi/EXAMPLES.md +0 -36
  116. package/.agents/core/skills/fastapi/SKILL.md +0 -148
  117. package/.agents/core/skills/fastapi/TROUBLESHOOTING.md +0 -19
  118. package/.agents/core/skills/fastapi/VALIDATION.json +0 -12
  119. package/.agents/core/skills/fastapi/fastapi.md +0 -112
  120. package/.agents/core/skills/fastapi/skill.yaml +0 -17
  121. package/.agents/core/skills/generators/EXAMPLES.md +0 -19
  122. package/.agents/core/skills/generators/SKILL.md +0 -112
  123. package/.agents/core/skills/generators/TROUBLESHOOTING.md +0 -7
  124. package/.agents/core/skills/generators/VALIDATION.json +0 -12
  125. package/.agents/core/skills/generators/skill.yaml +0 -25
  126. package/.agents/core/skills/generators/templates/API.md +0 -77
  127. package/.agents/core/skills/generators/templates/ARCHITECTURE.md +0 -70
  128. package/.agents/core/skills/generators/templates/DATABASE.md +0 -42
  129. package/.agents/core/skills/generators/templates/DECISION.md +0 -46
  130. package/.agents/core/skills/generators/templates/PRD.md +0 -67
  131. package/.agents/core/skills/generators/templates/PROJECT_GRAPH.md +0 -56
  132. package/.agents/core/skills/generators/templates/ROADMAP.md +0 -51
  133. package/.agents/core/skills/generators/templates/TASKS.md +0 -43
  134. package/.agents/core/skills/generators/templates/UI.md +0 -73
  135. package/.agents/core/skills/graphify/EXAMPLES.md +0 -73
  136. package/.agents/core/skills/graphify/SKILL.md +0 -130
  137. package/.agents/core/skills/graphify/VALIDATION.json +0 -12
  138. package/.agents/core/skills/graphify/skill.yaml +0 -18
  139. package/.agents/core/skills/impeccable-design/EXAMPLES.md +0 -26
  140. package/.agents/core/skills/impeccable-design/SKILL.md +0 -201
  141. package/.agents/core/skills/impeccable-design/TROUBLESHOOTING.md +0 -19
  142. package/.agents/core/skills/impeccable-design/VALIDATION.json +0 -12
  143. package/.agents/core/skills/impeccable-design/skill.yaml +0 -20
  144. package/.agents/core/skills/interview-me/SKILL.md +0 -97
  145. package/.agents/core/skills/interview-me/VALIDATION.json +0 -12
  146. package/.agents/core/skills/interview-me/skill.yaml +0 -12
  147. package/.agents/core/skills/microservices/EXAMPLES.md +0 -38
  148. package/.agents/core/skills/microservices/SKILL.md +0 -164
  149. package/.agents/core/skills/microservices/TROUBLESHOOTING.md +0 -19
  150. package/.agents/core/skills/microservices/VALIDATION.json +0 -12
  151. package/.agents/core/skills/microservices/microservices.md +0 -119
  152. package/.agents/core/skills/microservices/skill.yaml +0 -17
  153. package/.agents/core/skills/minimalist-design/SKILL.md +0 -113
  154. package/.agents/core/skills/minimalist-design/VALIDATION.json +0 -12
  155. package/.agents/core/skills/minimalist-design/skill.yaml +0 -12
  156. package/.agents/core/skills/nestjs/EXAMPLES.md +0 -40
  157. package/.agents/core/skills/nestjs/SKILL.md +0 -139
  158. package/.agents/core/skills/nestjs/TROUBLESHOOTING.md +0 -19
  159. package/.agents/core/skills/nestjs/VALIDATION.json +0 -12
  160. package/.agents/core/skills/nestjs/nestjs.md +0 -103
  161. package/.agents/core/skills/nestjs/skill.yaml +0 -17
  162. package/.agents/core/skills/nextjs/EXAMPLES.md +0 -40
  163. package/.agents/core/skills/nextjs/SKILL.md +0 -163
  164. package/.agents/core/skills/nextjs/TROUBLESHOOTING.md +0 -19
  165. package/.agents/core/skills/nextjs/VALIDATION.json +0 -12
  166. package/.agents/core/skills/nextjs/nextjs.md +0 -67
  167. package/.agents/core/skills/nextjs/skill.yaml +0 -17
  168. package/.agents/core/skills/node/EXAMPLES.md +0 -80
  169. package/.agents/core/skills/node/SKILL.md +0 -128
  170. package/.agents/core/skills/node/TROUBLESHOOTING.md +0 -19
  171. package/.agents/core/skills/node/VALIDATION.json +0 -12
  172. package/.agents/core/skills/node/node.md +0 -87
  173. package/.agents/core/skills/node/skill.yaml +0 -17
  174. package/.agents/core/skills/performance/EXAMPLES.md +0 -30
  175. package/.agents/core/skills/performance/SKILL.md +0 -75
  176. package/.agents/core/skills/performance/TROUBLESHOOTING.md +0 -19
  177. package/.agents/core/skills/performance/VALIDATION.json +0 -12
  178. package/.agents/core/skills/performance/performance.md +0 -52
  179. package/.agents/core/skills/performance/skill.yaml +0 -17
  180. package/.agents/core/skills/react/EXAMPLES.md +0 -79
  181. package/.agents/core/skills/react/SKILL.md +0 -132
  182. package/.agents/core/skills/react/TROUBLESHOOTING.md +0 -19
  183. package/.agents/core/skills/react/VALIDATION.json +0 -12
  184. package/.agents/core/skills/react/react.md +0 -93
  185. package/.agents/core/skills/react/skill.yaml +0 -17
  186. package/.agents/core/skills/react-best-practices/SKILL.md +0 -155
  187. package/.agents/core/skills/react-best-practices/VALIDATION.json +0 -12
  188. package/.agents/core/skills/react-best-practices/skill.yaml +0 -14
  189. package/.agents/core/skills/redesign-audit/SKILL.md +0 -117
  190. package/.agents/core/skills/redesign-audit/VALIDATION.json +0 -12
  191. package/.agents/core/skills/redesign-audit/skill.yaml +0 -12
  192. package/.agents/core/skills/soft-design/SKILL.md +0 -108
  193. package/.agents/core/skills/soft-design/VALIDATION.json +0 -12
  194. package/.agents/core/skills/soft-design/skill.yaml +0 -12
  195. package/.agents/core/skills/state-management/EXAMPLES.md +0 -56
  196. package/.agents/core/skills/state-management/SKILL.md +0 -48
  197. package/.agents/core/skills/state-management/TROUBLESHOOTING.md +0 -18
  198. package/.agents/core/skills/state-management/VALIDATION.json +0 -11
  199. package/.agents/core/skills/state-management/skill.yaml +0 -28
  200. package/.agents/core/skills/subagent-orchestrator/SKILL.md +0 -117
  201. package/.agents/core/skills/subagent-orchestrator/VALIDATION.json +0 -12
  202. package/.agents/core/skills/subagent-orchestrator/skill.yaml +0 -12
  203. package/.agents/core/skills/system-design/EXAMPLES.md +0 -75
  204. package/.agents/core/skills/system-design/SKILL.md +0 -419
  205. package/.agents/core/skills/system-design/TROUBLESHOOTING.md +0 -19
  206. package/.agents/core/skills/system-design/VALIDATION.json +0 -12
  207. package/.agents/core/skills/system-design/skill.yaml +0 -20
  208. package/.agents/core/skills/system-design/system-design.md +0 -112
  209. package/.agents/core/skills/testing/EXAMPLES.md +0 -71
  210. package/.agents/core/skills/testing/SKILL.md +0 -70
  211. package/.agents/core/skills/testing/TROUBLESHOOTING.md +0 -18
  212. package/.agents/core/skills/testing/VALIDATION.json +0 -11
  213. package/.agents/core/skills/testing/skill.yaml +0 -32
  214. package/.agents/core/skills/typescript/EXAMPLES.md +0 -64
  215. package/.agents/core/skills/typescript/SKILL.md +0 -112
  216. package/.agents/core/skills/typescript/TROUBLESHOOTING.md +0 -19
  217. package/.agents/core/skills/typescript/VALIDATION.json +0 -12
  218. package/.agents/core/skills/typescript/skill.yaml +0 -17
  219. package/.agents/core/skills/typescript/typescript.md +0 -71
  220. package/.agents/core/skills/ui-design/EXAMPLES.md +0 -21
  221. package/.agents/core/skills/ui-design/SKILL.md +0 -124
  222. package/.agents/core/skills/ui-design/TROUBLESHOOTING.md +0 -19
  223. package/.agents/core/skills/ui-design/VALIDATION.json +0 -12
  224. package/.agents/core/skills/ui-design/skill.yaml +0 -17
  225. package/.agents/core/skills/ui-design/ui.md +0 -88
  226. package/.agents/core/skills/ui-ux-pro/EXAMPLES.md +0 -62
  227. package/.agents/core/skills/ui-ux-pro/SKILL.md +0 -375
  228. package/.agents/core/skills/ui-ux-pro/TROUBLESHOOTING.md +0 -19
  229. package/.agents/core/skills/ui-ux-pro/VALIDATION.json +0 -12
  230. package/.agents/core/skills/ui-ux-pro/skill.yaml +0 -19
  231. package/.agents/core/skills/ux-design/EXAMPLES.md +0 -36
  232. package/.agents/core/skills/ux-design/SKILL.md +0 -116
  233. package/.agents/core/skills/ux-design/TROUBLESHOOTING.md +0 -19
  234. package/.agents/core/skills/ux-design/VALIDATION.json +0 -12
  235. package/.agents/core/skills/ux-design/skill.yaml +0 -17
  236. package/.agents/core/skills/ux-design/ux.md +0 -80
  237. package/.agents/core/skills/vercel-optimize/SKILL.md +0 -83
  238. package/.agents/core/skills/vercel-optimize/VALIDATION.json +0 -12
  239. package/.agents/core/skills/vercel-optimize/scripts/collect-signals.mjs +0 -131
  240. package/.agents/core/skills/vercel-optimize/scripts/gate-investigations.mjs +0 -142
  241. package/.agents/core/skills/vercel-optimize/scripts/merge-signals.mjs +0 -143
  242. package/.agents/core/skills/vercel-optimize/scripts/scan-codebase.mjs +0 -174
  243. package/.agents/core/skills/vercel-optimize/skill.yaml +0 -18
  244. package/.agents/core/skills/web-accessibility/EXAMPLES.md +0 -39
  245. package/.agents/core/skills/web-accessibility/SKILL.md +0 -170
  246. package/.agents/core/skills/web-accessibility/TROUBLESHOOTING.md +0 -19
  247. package/.agents/core/skills/web-accessibility/VALIDATION.json +0 -12
  248. package/.agents/core/skills/web-accessibility/accessibility.md +0 -63
  249. package/.agents/core/skills/web-accessibility/skill.yaml +0 -17
  250. package/.agents/generated/claude/skills/adapters/SKILL.md +0 -126
  251. package/.agents/generated/claude/skills/architecture-diagrams/SKILL.md +0 -101
  252. package/.agents/generated/claude/skills/brutalist-design/SKILL.md +0 -145
  253. package/.agents/generated/claude/skills/database/SKILL.md +0 -191
  254. package/.agents/generated/claude/skills/ddd/SKILL.md +0 -305
  255. package/.agents/generated/claude/skills/decisions/SKILL.md +0 -134
  256. package/.agents/generated/claude/skills/docker/SKILL.md +0 -135
  257. package/.agents/generated/claude/skills/fastapi/SKILL.md +0 -200
  258. package/.agents/generated/claude/skills/generators/SKILL.md +0 -133
  259. package/.agents/generated/claude/skills/graphify/SKILL.md +0 -198
  260. package/.agents/generated/claude/skills/impeccable-design/SKILL.md +0 -241
  261. package/.agents/generated/claude/skills/interview-me/SKILL.md +0 -90
  262. package/.agents/generated/claude/skills/microservices/SKILL.md +0 -218
  263. package/.agents/generated/claude/skills/minimalist-design/SKILL.md +0 -108
  264. package/.agents/generated/claude/skills/nestjs/SKILL.md +0 -195
  265. package/.agents/generated/claude/skills/nextjs/SKILL.md +0 -219
  266. package/.agents/generated/claude/skills/node/SKILL.md +0 -224
  267. package/.agents/generated/claude/skills/performance/SKILL.md +0 -121
  268. package/.agents/generated/claude/skills/react/SKILL.md +0 -227
  269. package/.agents/generated/claude/skills/react-best-practices/SKILL.md +0 -146
  270. package/.agents/generated/claude/skills/redesign-audit/SKILL.md +0 -112
  271. package/.agents/generated/claude/skills/soft-design/SKILL.md +0 -103
  272. package/.agents/generated/claude/skills/state-management/SKILL.md +0 -120
  273. package/.agents/generated/claude/skills/subagent-orchestrator/SKILL.md +0 -110
  274. package/.agents/generated/claude/skills/system-design/SKILL.md +0 -507
  275. package/.agents/generated/claude/skills/testing/SKILL.md +0 -157
  276. package/.agents/generated/claude/skills/typescript/SKILL.md +0 -192
  277. package/.agents/generated/claude/skills/ui-design/SKILL.md +0 -161
  278. package/.agents/generated/claude/skills/ui-ux-pro/SKILL.md +0 -451
  279. package/.agents/generated/claude/skills/ux-design/SKILL.md +0 -168
  280. package/.agents/generated/claude/skills/vercel-optimize/SKILL.md +0 -76
  281. package/.agents/generated/claude/skills/web-accessibility/SKILL.md +0 -225
  282. package/.agents/generated/gemini/skills/adapters/SKILL.md +0 -135
  283. package/.agents/generated/gemini/skills/architecture-diagrams/SKILL.md +0 -107
  284. package/.agents/generated/gemini/skills/brutalist-design/SKILL.md +0 -151
  285. package/.agents/generated/gemini/skills/database/SKILL.md +0 -200
  286. package/.agents/generated/gemini/skills/ddd/SKILL.md +0 -314
  287. package/.agents/generated/gemini/skills/decisions/SKILL.md +0 -143
  288. package/.agents/generated/gemini/skills/docker/SKILL.md +0 -144
  289. package/.agents/generated/gemini/skills/fastapi/SKILL.md +0 -209
  290. package/.agents/generated/gemini/skills/generators/SKILL.md +0 -142
  291. package/.agents/generated/gemini/skills/graphify/SKILL.md +0 -205
  292. package/.agents/generated/gemini/skills/impeccable-design/SKILL.md +0 -250
  293. package/.agents/generated/gemini/skills/interview-me/SKILL.md +0 -96
  294. package/.agents/generated/gemini/skills/microservices/SKILL.md +0 -227
  295. package/.agents/generated/gemini/skills/minimalist-design/SKILL.md +0 -114
  296. package/.agents/generated/gemini/skills/nestjs/SKILL.md +0 -204
  297. package/.agents/generated/gemini/skills/nextjs/SKILL.md +0 -298
  298. package/.agents/generated/gemini/skills/node/SKILL.md +0 -323
  299. package/.agents/generated/gemini/skills/performance/SKILL.md +0 -185
  300. package/.agents/generated/gemini/skills/react/SKILL.md +0 -332
  301. package/.agents/generated/gemini/skills/react-best-practices/SKILL.md +0 -152
  302. package/.agents/generated/gemini/skills/redesign-audit/SKILL.md +0 -118
  303. package/.agents/generated/gemini/skills/soft-design/SKILL.md +0 -109
  304. package/.agents/generated/gemini/skills/state-management/SKILL.md +0 -129
  305. package/.agents/generated/gemini/skills/subagent-orchestrator/SKILL.md +0 -116
  306. package/.agents/generated/gemini/skills/system-design/SKILL.md +0 -631
  307. package/.agents/generated/gemini/skills/testing/SKILL.md +0 -166
  308. package/.agents/generated/gemini/skills/typescript/SKILL.md +0 -275
  309. package/.agents/generated/gemini/skills/ui-design/SKILL.md +0 -170
  310. package/.agents/generated/gemini/skills/ui-ux-pro/SKILL.md +0 -460
  311. package/.agents/generated/gemini/skills/ux-design/SKILL.md +0 -177
  312. package/.agents/generated/gemini/skills/vercel-optimize/SKILL.md +0 -82
  313. package/.agents/generated/gemini/skills/web-accessibility/SKILL.md +0 -300
@@ -0,0 +1,140 @@
1
+ /**
2
+ * benchmarks/v2/analysis/statistics.js
3
+ * ContextOS Benchmark v2 — Statistical Analysis Engine
4
+ *
5
+ * Implements Section 24 of CONTEXTOS_IMPLEMENTATION_PLAN.md:
6
+ * - Primary metric: cost per independently verified successful task
7
+ * - Wilson score 95% confidence intervals for binomial success proportions
8
+ * - Pairwise delta comparison between Arm C/D and Arm B (Concise Checklist comparator)
9
+ * - Structured tabular summary generation
10
+ */
11
+
12
+ 'use strict';
13
+
14
+ /**
15
+ * Calculates Wilson score 95% confidence interval for a proportion.
16
+ *
17
+ * @param {number} successes
18
+ * @param {number} total
19
+ * @param {number} [z=1.96] - 95% confidence z-score
20
+ * @returns {[number, number]} [lower, upper] as percentages [0, 100]
21
+ */
22
+ function calculateWilsonInterval(successes, total, z = 1.96) {
23
+ if (total === 0) return [0, 0];
24
+ const p = successes / total;
25
+ const z2 = z * z;
26
+ const denominator = 1 + z2 / total;
27
+ const center = (p + z2 / (2 * total)) / denominator;
28
+ const margin = (z * Math.sqrt((p * (1 - p)) / total + z2 / (4 * total * total))) / denominator;
29
+
30
+ return [
31
+ Math.max(0, parseFloat(((center - margin) * 100).toFixed(1))),
32
+ Math.min(100, parseFloat(((center + margin) * 100).toFixed(1))),
33
+ ];
34
+ }
35
+
36
+ class BenchmarkStatistics {
37
+ /**
38
+ * Analyzes an array of run outcomes across experimental arms.
39
+ *
40
+ * @param {Array<Object>} runs - List of run records: { armId, success, totalCost, durationMs }
41
+ * @returns {Object} Comprehensive statistical summary
42
+ */
43
+ static analyze(runs) {
44
+ const armsMap = {};
45
+
46
+ for (const run of runs) {
47
+ const { armId, success, totalCost = 0, durationMs = 0 } = run;
48
+ if (!armsMap[armId]) {
49
+ armsMap[armId] = {
50
+ armId,
51
+ totalRuns: 0,
52
+ successfulRuns: 0,
53
+ totalCost: 0,
54
+ totalDurationMs: 0,
55
+ };
56
+ }
57
+
58
+ const item = armsMap[armId];
59
+ item.totalRuns++;
60
+ if (success) item.successfulRuns++;
61
+ item.totalCost += totalCost;
62
+ item.totalDurationMs += durationMs;
63
+ }
64
+
65
+ const armStats = {};
66
+ for (const [armId, d] of Object.entries(armsMap)) {
67
+ const successRate = d.totalRuns > 0 ? (d.successfulRuns / d.totalRuns) * 100 : 0;
68
+ const ci95 = calculateWilsonInterval(d.successfulRuns, d.totalRuns);
69
+ const costPerSuccess = d.successfulRuns > 0 ? d.totalCost / d.successfulRuns : null;
70
+ const avgDurationMs = d.totalRuns > 0 ? d.totalDurationMs / d.totalRuns : 0;
71
+
72
+ armStats[armId] = {
73
+ armId,
74
+ totalRuns: d.totalRuns,
75
+ successfulRuns: d.successfulRuns,
76
+ successRate: parseFloat(successRate.toFixed(1)),
77
+ ci95,
78
+ totalCost: parseFloat(d.totalCost.toFixed(4)),
79
+ costPerVerifiedSuccess: costPerSuccess !== null ? parseFloat(costPerSuccess.toFixed(4)) : null,
80
+ avgDurationMs: Math.round(avgDurationMs),
81
+ };
82
+ }
83
+
84
+ // Pairwise comparisons against Arm B (Concise Checklist)
85
+ const comparator = armStats['arm-b-concise-checklist'];
86
+ const comparisons = {};
87
+
88
+ if (comparator) {
89
+ for (const [armId, stat] of Object.entries(armStats)) {
90
+ if (armId === 'arm-b-concise-checklist') continue;
91
+
92
+ const rateDelta = parseFloat((stat.successRate - comparator.successRate).toFixed(1));
93
+ let costRatio = null;
94
+ if (comparator.costPerVerifiedSuccess && stat.costPerVerifiedSuccess) {
95
+ costRatio = parseFloat((stat.costPerVerifiedSuccess / comparator.costPerVerifiedSuccess).toFixed(2));
96
+ }
97
+
98
+ comparisons[armId] = {
99
+ vsComparator: 'arm-b-concise-checklist',
100
+ successRateDelta: rateDelta,
101
+ costRatio,
102
+ };
103
+ }
104
+ }
105
+
106
+ return {
107
+ timestamp: Date.now(),
108
+ totalRunsAnalyzed: runs.length,
109
+ arms: armStats,
110
+ comparisons,
111
+ };
112
+ }
113
+
114
+ /**
115
+ * Formats statistical analysis into an aligned markdown summary table.
116
+ *
117
+ * @param {Object} analysis
118
+ * @returns {string}
119
+ */
120
+ static formatTable(analysis) {
121
+ const lines = [];
122
+ lines.push('| Arm ID | Runs | Successes | Success Rate (95% CI) | Cost / Verified Success | Avg Latency |');
123
+ lines.push('|---|---:|---:|---:|---:|---:|');
124
+
125
+ for (const arm of Object.values(analysis.arms)) {
126
+ const ciStr = `[${arm.ci95[0]}%, ${arm.ci95[1]}%]`;
127
+ const costStr = arm.costPerVerifiedSuccess !== null ? `$${arm.costPerVerifiedSuccess}` : 'N/A';
128
+ lines.push(
129
+ `| **${arm.armId}** | ${arm.totalRuns} | ${arm.successfulRuns} | ${arm.successRate}% ${ciStr} | ${costStr} | ${arm.avgDurationMs}ms |`
130
+ );
131
+ }
132
+
133
+ return lines.join('\n');
134
+ }
135
+ }
136
+
137
+ module.exports = {
138
+ calculateWilsonInterval,
139
+ BenchmarkStatistics,
140
+ };
@@ -0,0 +1,69 @@
1
+ 'use strict';
2
+
3
+ /**
4
+ * Calculates 95% Confidence Interval for a proportion using the Wald method.
5
+ * @param {number} p - sample proportion (success rate)
6
+ * @param {number} n - sample size
7
+ * @returns {number} Margin of Error
8
+ */
9
+ function calculate95CI(p, n) {
10
+ if (n === 0) return 0;
11
+ // Z-value for 95% confidence is 1.96
12
+ const z = 1.96;
13
+ const standardError = Math.sqrt((p * (1 - p)) / n);
14
+ return z * standardError;
15
+ }
16
+
17
+ /**
18
+ * Analyzes the results of a benchmark run.
19
+ * @param {Array} results Array of execution result objects from the runner
20
+ * @returns {Object} Statistical summary
21
+ */
22
+ function analyzeResults(results) {
23
+ const statsByArm = {};
24
+
25
+ // Group by arm
26
+ for (const res of results) {
27
+ if (!statsByArm[res.armId]) {
28
+ statsByArm[res.armId] = {
29
+ total: 0,
30
+ successes: 0,
31
+ failures: 0,
32
+ totalTokens: 0,
33
+ totalDurationMs: 0
34
+ };
35
+ }
36
+
37
+ const stats = statsByArm[res.armId];
38
+ stats.total++;
39
+ if (res.success) {
40
+ stats.successes++;
41
+ } else {
42
+ stats.failures++;
43
+ }
44
+ stats.totalTokens += res.usage.totalTokens;
45
+ stats.totalDurationMs += res.durationMs;
46
+ }
47
+
48
+ // Calculate rates and CI
49
+ const finalStats = {};
50
+ for (const [armId, stats] of Object.entries(statsByArm)) {
51
+ const successRate = stats.total > 0 ? stats.successes / stats.total : 0;
52
+ const marginOfError = calculate95CI(successRate, stats.total);
53
+
54
+ finalStats[armId] = {
55
+ ...stats,
56
+ successRate: parseFloat((successRate * 100).toFixed(2)),
57
+ confidenceInterval95: `±${(marginOfError * 100).toFixed(2)}%`,
58
+ avgTokensPerTask: stats.total > 0 ? Math.round(stats.totalTokens / stats.total) : 0,
59
+ avgDurationMs: stats.total > 0 ? Math.round(stats.totalDurationMs / stats.total) : 0
60
+ };
61
+ }
62
+
63
+ return finalStats;
64
+ }
65
+
66
+ module.exports = {
67
+ analyzeResults,
68
+ calculate95CI
69
+ };
@@ -0,0 +1,79 @@
1
+ /**
2
+ * benchmarks/v2/arms/arm-definitions.js
3
+ * ContextOS Benchmark v2 — Evaluation Arms Specification
4
+ *
5
+ * Implements Section 24.2 of CONTEXTOS_IMPLEMENTATION_PLAN.md:
6
+ * - Arm A (Vanilla): Neutral baseline system prompt without artificial debuffing
7
+ * - Arm B (Concise Checklist): 10-15 universal engineering rules (~600 tokens) — Primary Comparator
8
+ * - Arm C (ContextOS Core): Dynamic canonical skill resolver without ceremony
9
+ * - Arm D (Full ContextOS): Resolver + risk workflow + isolated runtime verification + review
10
+ */
11
+
12
+ 'use strict';
13
+
14
+ const ARMS = {
15
+ ARM_A_VANILLA: {
16
+ id: 'arm-a-vanilla',
17
+ name: 'Vanilla Baseline',
18
+ description: 'Neutral baseline system prompt without ContextOS rules or checklists.',
19
+ tokenBudgetEstimate: 120,
20
+ buildSystemPrompt: () => {
21
+ return 'You are an expert software engineer. Write clean, complete, working production code that solves the user request.';
22
+ },
23
+ },
24
+
25
+ ARM_B_CONCISE_CHECKLIST: {
26
+ id: 'arm-b-concise-checklist',
27
+ name: 'Concise Checklist',
28
+ description: 'High-density 12-rule engineering checklist (~600 tokens). Primary comparator.',
29
+ tokenBudgetEstimate: 580,
30
+ buildSystemPrompt: () => {
31
+ return [
32
+ 'You are a Senior Staff Engineer.',
33
+ 'Follow this strict engineering checklist:',
34
+ '1. Inspect existing files before editing.',
35
+ '2. Never use placeholders, stubs, or TODO comments.',
36
+ '3. Maintain existing codebase naming conventions and architectural boundaries.',
37
+ '4. Minimize blast radius — modify only files required for the task.',
38
+ '5. Validate all user input and sanitize data paths.',
39
+ '6. Use parameterized queries for database operations.',
40
+ '7. Handle all asynchronous error boundaries explicitly.',
41
+ '8. Write comprehensive unit and integration test assertions.',
42
+ '9. Ensure clean TypeScript typing without any unsafe casts.',
43
+ '10. Verify backward compatibility with existing public APIs.',
44
+ '11. No secrets or credentials in code or commits.',
45
+ '12. Ensure code compiles and all tests pass.',
46
+ ].join('\n');
47
+ },
48
+ },
49
+
50
+ ARM_C_CONTEXTOS_CORE: {
51
+ id: 'arm-c-contextos-core',
52
+ name: 'ContextOS Core (Dynamic Context Selection)',
53
+ description: 'Dynamic canonical resolver selecting exact skills and rules without ceremony.',
54
+ tokenBudgetEstimate: 1400,
55
+ buildSystemPrompt: (resolvedSkills = []) => {
56
+ const skillsHeader = resolvedSkills.length > 0
57
+ ? `[ContextOS Resolved Skills: ${resolvedSkills.join(', ')}]`
58
+ : '[ContextOS Core]';
59
+ return `${skillsHeader}\nExecute task adhering to compiled workspace rules and exact skill invariants.`;
60
+ },
61
+ },
62
+
63
+ ARM_D_FULL_CONTEXTOS: {
64
+ id: 'arm-d-full-contextos',
65
+ name: 'Full ContextOS (Core + Runtime Verification)',
66
+ description: 'Dynamic resolver + risk-based workflow + isolated runtime verification + reviewer pipeline.',
67
+ tokenBudgetEstimate: 2200,
68
+ buildSystemPrompt: (resolvedSkills = [], riskLevel = 'STANDARD') => {
69
+ return [
70
+ `[ContextOS Full Runtime] [RISK: ${riskLevel}] [Skills: ${resolvedSkills.join(', ')}]`,
71
+ 'Execution gated by isolated worktree and mandatory verification attestations before merge readiness.',
72
+ ].join('\n');
73
+ },
74
+ },
75
+ };
76
+
77
+ module.exports = {
78
+ ARMS,
79
+ };
@@ -0,0 +1,34 @@
1
+ {
2
+ "$schema": "http://json-schema.org/draft-07/schema#",
3
+ "title": "ContextOS Benchmark v2 Task Schema",
4
+ "type": "object",
5
+ "properties": {
6
+ "id": {
7
+ "type": "string",
8
+ "description": "Unique immutable identifier for the task"
9
+ },
10
+ "description": {
11
+ "type": "string",
12
+ "description": "The actual prompt provided to the LLM"
13
+ },
14
+ "expectedState": {
15
+ "type": "object",
16
+ "description": "The expected state of the filesystem or output after execution",
17
+ "properties": {
18
+ "filesToExist": {
19
+ "type": "array",
20
+ "items": { "type": "string" }
21
+ },
22
+ "filesToContain": {
23
+ "type": "object",
24
+ "additionalProperties": { "type": "string" }
25
+ }
26
+ }
27
+ },
28
+ "hash": {
29
+ "type": "string",
30
+ "description": "SHA-256 hash of the task for immutability verification"
31
+ }
32
+ },
33
+ "required": ["id", "description", "hash"]
34
+ }
@@ -0,0 +1,25 @@
1
+ 'use strict';
2
+
3
+ /**
4
+ * Evaluates the result of a task execution against the expected state.
5
+ * Since we are using a Mock Provider, the evaluation logic is simplified
6
+ * to just trust the provider's mocked success status. In a real system,
7
+ * this would run ESLint, execute the generated code in a sandbox,
8
+ * and verify the exact AST or output.
9
+ *
10
+ * @param {Object} task The benchmark task definition
11
+ * @param {Object} llmResult The result from the LLM/Mock Provider
12
+ * @returns {Object} { passed: boolean, error: string|null }
13
+ */
14
+ function evaluateTask(task, llmResult) {
15
+ if (llmResult.success) {
16
+ return { passed: true, error: null };
17
+ } else {
18
+ return {
19
+ passed: false,
20
+ error: `Failed to meet expected state for task ${task.id}: ${llmResult.mockedOutput}`
21
+ };
22
+ }
23
+ }
24
+
25
+ module.exports = evaluateTask;
@@ -0,0 +1,116 @@
1
+ /**
2
+ * benchmarks/v2/evaluators/verified-success.js
3
+ * ContextOS Benchmark v2 — Primary Outcome Evaluator
4
+ *
5
+ * Implements Section 24.8 of CONTEXTOS_IMPLEMENTATION_PLAN.md:
6
+ * - Primary metric: independently_verified_success
7
+ * - Evaluates:
8
+ * 1. Patch application / syntax correctness
9
+ * 2. Typecheck / build status
10
+ * 3. Public unit test pass rate
11
+ * 4. Hidden test suite pass rate (isolated oracle)
12
+ * 5. Zero regression on baseline suites
13
+ * 6. Zero P0/P1 security findings or leaked credentials
14
+ * 7. Zero placeholder stubs (TODO, mock placeholders)
15
+ * 8. Budget constraints (tokens, time, turns)
16
+ */
17
+
18
+ 'use strict';
19
+
20
+ const PLACEHOLDER_PATTERNS = [
21
+ /\/\/\s*TODO:\s*implement\b/i,
22
+ /\/\/\s*\.\.\.\s*rest of code\b/i,
23
+ /\bthrow new Error\(["']Not implemented["']\)/i,
24
+ /\bpass\s*#\s*TODO\b/i,
25
+ ];
26
+
27
+ class BenchmarkEvaluator {
28
+ /**
29
+ * Evaluates task run evidence against rigorous quality gates.
30
+ *
31
+ * @param {Object} runEvidence
32
+ * @param {boolean} runEvidence.patchApplied
33
+ * @param {boolean} runEvidence.buildPass
34
+ * @param {number} runEvidence.publicTestsTotal
35
+ * @param {number} runEvidence.publicTestsPassed
36
+ * @param {number} runEvidence.hiddenTestsTotal
37
+ * @param {number} runEvidence.hiddenTestsPassed
38
+ * @param {boolean} [runEvidence.regressions=false]
39
+ * @param {Array<string>} [runEvidence.securityFindings=[]]
40
+ * @param {string} [runEvidence.generatedCode='']
41
+ * @param {Object} [runEvidence.budget]
42
+ * @param {number} [runEvidence.budget.tokensUsed=0]
43
+ * @param {number} [runEvidence.budget.tokenLimit=50000]
44
+ * @param {number} [runEvidence.budget.durationMs=0]
45
+ * @param {number} [runEvidence.budget.timeoutMs=60000]
46
+ * @returns {Object} Evaluation report
47
+ */
48
+ static evaluate(runEvidence) {
49
+ const {
50
+ patchApplied = false,
51
+ buildPass = false,
52
+ publicTestsTotal = 0,
53
+ publicTestsPassed = 0,
54
+ hiddenTestsTotal = 0,
55
+ hiddenTestsPassed = 0,
56
+ regressions = false,
57
+ securityFindings = [],
58
+ generatedCode = '',
59
+ budget = {},
60
+ } = runEvidence;
61
+
62
+ const failures = [];
63
+
64
+ // 1. Patch & build
65
+ if (!patchApplied) failures.push('Patch was not successfully applied');
66
+ if (!buildPass) failures.push('Compilation or typecheck failed');
67
+
68
+ // 2. Tests
69
+ if (publicTestsTotal > 0 && publicTestsPassed < publicTestsTotal) {
70
+ failures.push(`Public tests failed: ${publicTestsPassed}/${publicTestsTotal}`);
71
+ }
72
+ if (hiddenTestsTotal > 0 && hiddenTestsPassed < hiddenTestsTotal) {
73
+ failures.push(`Hidden test oracle failed: ${hiddenTestsPassed}/${hiddenTestsTotal}`);
74
+ }
75
+ if (regressions) {
76
+ failures.push('Regression detected in existing test baseline');
77
+ }
78
+
79
+ // 3. Security
80
+ if (Array.isArray(securityFindings) && securityFindings.length > 0) {
81
+ failures.push(`Security vulnerabilities detected: ${securityFindings.join(', ')}`);
82
+ }
83
+
84
+ // 4. Zero placeholders
85
+ if (generatedCode) {
86
+ for (const pat of PLACEHOLDER_PATTERNS) {
87
+ if (pat.test(generatedCode)) {
88
+ failures.push(`Lazy placeholder detected matching pattern: ${pat.source}`);
89
+ break;
90
+ }
91
+ }
92
+ }
93
+
94
+ // 5. Budget constraints
95
+ if (budget.tokensUsed && budget.tokenLimit && budget.tokensUsed > budget.tokenLimit) {
96
+ failures.push(`Token budget exceeded: ${budget.tokensUsed} > ${budget.tokenLimit}`);
97
+ }
98
+ if (budget.durationMs && budget.timeoutMs && budget.durationMs > budget.timeoutMs) {
99
+ failures.push(`Time budget exceeded: ${budget.durationMs}ms > ${budget.timeoutMs}ms`);
100
+ }
101
+
102
+ const isSuccess = failures.length === 0;
103
+
104
+ return {
105
+ independently_verified_success: isSuccess,
106
+ publicTestRate: publicTestsTotal > 0 ? publicTestsPassed / publicTestsTotal : 1.0,
107
+ hiddenTestRate: hiddenTestsTotal > 0 ? hiddenTestsPassed / hiddenTestsTotal : 1.0,
108
+ failureCount: failures.length,
109
+ failures,
110
+ };
111
+ }
112
+ }
113
+
114
+ module.exports = {
115
+ BenchmarkEvaluator,
116
+ };
@@ -0,0 +1,88 @@
1
+ 'use strict';
2
+
3
+ const crypto = require('crypto');
4
+ const { ARMS } = require('../arms/arm-definitions');
5
+ const evaluateTask = require('../evaluators/index');
6
+
7
+ /**
8
+ * Mock LLM Provider used when real API keys are unavailable.
9
+ * Deterministically simulates success/failure rates based on the arm's capability.
10
+ */
11
+ class MockProvider {
12
+ /**
13
+ * Probability of success for each arm to simulate real-world capability differences.
14
+ */
15
+ static getSuccessProbability(armId) {
16
+ switch (armId) {
17
+ case ARMS.ARM_A_VANILLA.id: return 0.40; // 40% success
18
+ case ARMS.ARM_B_CONCISE_CHECKLIST.id: return 0.65; // 65% success
19
+ case ARMS.ARM_C_CONTEXTOS_CORE.id: return 0.85; // 85% success
20
+ case ARMS.ARM_D_FULL_CONTEXTOS.id: return 0.98; // 98% success
21
+ default: return 0.0;
22
+ }
23
+ }
24
+
25
+ static async execute(task, arm) {
26
+ const probability = this.getSuccessProbability(arm.id);
27
+ // Use hash to deterministically seed pseudo-randomness for the mock run
28
+ const hashInt = parseInt(task.hash.substring(7, 15), 16);
29
+ const successThreshold = probability * 0xffffffff;
30
+
31
+ // Slight artificial delay to simulate API request
32
+ await new Promise(resolve => setTimeout(resolve, 50));
33
+
34
+ const isSuccess = hashInt <= successThreshold;
35
+
36
+ return {
37
+ success: isSuccess,
38
+ mockedOutput: isSuccess
39
+ ? `Successfully generated code for ${task.id}`
40
+ : `Failed or hallucinated output for ${task.id}`,
41
+ usage: {
42
+ promptTokens: arm.tokenBudgetEstimate,
43
+ completionTokens: 300,
44
+ totalTokens: arm.tokenBudgetEstimate + 300
45
+ }
46
+ };
47
+ }
48
+ }
49
+
50
+ /**
51
+ * Executes a single task against a single arm.
52
+ */
53
+ async function runTask(task, armId) {
54
+ const arm = Object.values(ARMS).find(a => a.id === armId);
55
+ if (!arm) throw new Error(`Unknown arm: ${armId}`);
56
+
57
+ const startTime = Date.now();
58
+ const requestId = crypto.randomUUID();
59
+
60
+ // Execute using Mock Provider (replace with real LLM client when API keys are available)
61
+ const llmResult = await MockProvider.execute(task, arm);
62
+
63
+ const durationMs = Date.now() - startTime;
64
+
65
+ // Evaluate the output
66
+ const evaluation = evaluateTask(task, llmResult);
67
+
68
+ return {
69
+ taskId: task.id,
70
+ armId: arm.id,
71
+ requestId,
72
+ timestamp: new Date().toISOString(),
73
+ durationMs,
74
+ success: evaluation.passed,
75
+ error: evaluation.error || null,
76
+ usage: llmResult.usage,
77
+ environmentProvenance: {
78
+ nodeVersion: process.version,
79
+ platform: process.platform,
80
+ engine: 'mock-provider-v1'
81
+ }
82
+ };
83
+ }
84
+
85
+ module.exports = {
86
+ runTask,
87
+ MockProvider
88
+ };