contextos-agents 1.6.1 → 2.0.0-beta.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (323) hide show
  1. package/.agents/AGENTS.md +6 -1
  2. package/.agents/adapters/aider/export.js +117 -97
  3. package/.agents/adapters/claude/export.js +68 -26
  4. package/.agents/adapters/copilot/export.js +90 -51
  5. package/.agents/adapters/cursor/export.js +83 -68
  6. package/.agents/adapters/drift-detector.js +196 -0
  7. package/.agents/adapters/gemini/export.js +76 -45
  8. package/.agents/adapters/pure-compiler.js +443 -0
  9. package/.agents/adapters/zed/export.js +109 -62
  10. package/.agents/compiled/registry.v2.json +504 -0
  11. package/.agents/compiled/registry.v2.sha256 +1 -0
  12. package/.agents/compiler/manifest-compiler.js +963 -0
  13. package/.agents/compiler/vendor/yaml.LICENSE.txt +13 -0
  14. package/.agents/compiler/vendor/yaml.SBOM.json +6 -0
  15. package/.agents/compiler/vendor/yaml.js +139 -0
  16. package/.agents/core/profiles/init.yaml +25 -0
  17. package/.agents/core/skills/context-manager/references/context-rules.md +59 -0
  18. package/.agents/core/skills/context-manager/skill.yaml +10 -5
  19. package/.agents/core/skills/context-os/SKILL.md +3 -6
  20. package/.agents/core/skills/context-os/skill.yaml +14 -8
  21. package/.agents/core/skills/engineering-workflow/SKILL.md +1 -1
  22. package/.agents/core/skills/engineering-workflow/skill.yaml +7 -7
  23. package/.agents/core/skills/gemini-precision/SKILL.md +4 -0
  24. package/.agents/core/skills/gemini-precision/skill.yaml +5 -6
  25. package/.agents/core/skills/gstack-roles/SKILL.md +3 -1
  26. package/.agents/core/skills/gstack-roles/skill.yaml +9 -6
  27. package/.agents/core/skills/ponytail-mindset/skill.yaml +7 -7
  28. package/.agents/core/skills/security/skill.yaml +21 -2
  29. package/.agents/ctx.js +587 -111
  30. package/.agents/customization-dx.js +282 -0
  31. package/.agents/doctor.js +877 -33
  32. package/.agents/filesystem/index.js +71 -0
  33. package/.agents/filesystem/journaled-transaction.js +451 -0
  34. package/.agents/filesystem/lockfile-v2.js +275 -0
  35. package/.agents/filesystem/platform-hardening.js +222 -0
  36. package/.agents/filesystem/project-lock.js +218 -0
  37. package/.agents/filesystem/safe-path.js +256 -0
  38. package/.agents/generated/claude/skills/context-os/SKILL.md +1 -1
  39. package/.agents/generated/claude/skills/engineering-workflow/SKILL.md +1 -1
  40. package/.agents/generated/claude/skills/gemini-precision/SKILL.md +4 -0
  41. package/.agents/generated/claude/skills/gstack-roles/SKILL.md +3 -1
  42. package/.agents/generated/gemini/skills/context-os/SKILL.md +2 -2
  43. package/.agents/generated/gemini/skills/engineering-workflow/SKILL.md +1 -1
  44. package/.agents/generated/gemini/skills/gemini-precision/SKILL.md +4 -0
  45. package/.agents/generated/gemini/skills/gstack-roles/SKILL.md +3 -1
  46. package/.agents/plugins/contextos/hooks.json +25 -0
  47. package/.agents/plugins/contextos/plugin.json +19 -0
  48. package/.agents/plugins.js +432 -73
  49. package/.agents/profiles.js +507 -46
  50. package/.agents/resolver.js +50 -414
  51. package/.agents/schemas/attestation.review.v1.json +111 -0
  52. package/.agents/schemas/attestation.verification.v1.json +85 -0
  53. package/.agents/schemas/lockfile.v2.schema.json +134 -0
  54. package/.agents/schemas/profile.v2.schema.json +114 -0
  55. package/.agents/schemas/runtime.thread.v1.json +192 -0
  56. package/.agents/schemas/skill.manifest.v2.json +177 -0
  57. package/.agents/schemas/verification.spec.v1.json +39 -0
  58. package/.agents/schemas/workspace.graph.schema.json +106 -0
  59. package/.agents/stats.js +22 -9
  60. package/.agents/transaction-core/event-store.js +288 -0
  61. package/.agents/transaction-core/idempotency.js +129 -0
  62. package/.agents/transaction-core/ipc-lock.js +311 -0
  63. package/.agents/transaction-core/plugin-supply-chain-bundle.js +436 -0
  64. package/.agents/validate.js +143 -14
  65. package/.agents/watch.js +354 -102
  66. package/.agents/workspace/workspace-graph.js +778 -0
  67. package/README.md +59 -387
  68. package/benchmarks/v2/analysis/statistics.js +140 -0
  69. package/benchmarks/v2/analysis/stats.js +69 -0
  70. package/benchmarks/v2/arms/arm-definitions.js +79 -0
  71. package/benchmarks/v2/dataset.schema.json +34 -0
  72. package/benchmarks/v2/evaluators/index.js +25 -0
  73. package/benchmarks/v2/evaluators/verified-success.js +116 -0
  74. package/benchmarks/v2/harness/runner.js +88 -0
  75. package/benchmarks/v2/pilot-tasks.json +392 -0
  76. package/bin/commands/recover.js +88 -0
  77. package/bin/commands/uninstall.js +207 -0
  78. package/bin/commands/update.js +325 -0
  79. package/bin/commands.js +342 -0
  80. package/bin/index.js +326 -149
  81. package/bin/lib/detector.js +106 -0
  82. package/bin/lib/lockfile.js +253 -0
  83. package/bin/lib/safe-writer.js +290 -0
  84. package/package.json +85 -73
  85. package/registry.json +15 -7
  86. package/registry.schema.json +3 -1
  87. package/registry.v2.schema.json +86 -0
  88. package/.agents/core/profiles/backend.yaml +0 -47
  89. package/.agents/core/profiles/enterprise.yaml +0 -46
  90. package/.agents/core/profiles/frontend.yaml +0 -46
  91. package/.agents/core/profiles/hackathon.yaml +0 -45
  92. package/.agents/core/profiles/mvp.yaml +0 -44
  93. package/.agents/core/profiles/startup.yaml +0 -48
  94. package/.agents/core/skills/adapters/EXAMPLES.md +0 -19
  95. package/.agents/core/skills/adapters/SKILL.md +0 -105
  96. package/.agents/core/skills/adapters/TROUBLESHOOTING.md +0 -7
  97. package/.agents/core/skills/adapters/VALIDATION.json +0 -12
  98. package/.agents/core/skills/adapters/skill.yaml +0 -10
  99. package/.agents/core/skills/architecture-diagrams/SKILL.md +0 -108
  100. package/.agents/core/skills/architecture-diagrams/VALIDATION.json +0 -12
  101. package/.agents/core/skills/architecture-diagrams/skill.yaml +0 -8
  102. package/.agents/core/skills/brutalist-design/SKILL.md +0 -150
  103. package/.agents/core/skills/brutalist-design/VALIDATION.json +0 -12
  104. package/.agents/core/skills/brutalist-design/skill.yaml +0 -8
  105. package/.agents/core/skills/database/EXAMPLES.md +0 -74
  106. package/.agents/core/skills/database/SKILL.md +0 -101
  107. package/.agents/core/skills/database/TROUBLESHOOTING.md +0 -18
  108. package/.agents/core/skills/database/VALIDATION.json +0 -11
  109. package/.agents/core/skills/database/skill.yaml +0 -25
  110. package/.agents/core/skills/ddd/EXAMPLES.md +0 -42
  111. package/.agents/core/skills/ddd/SKILL.md +0 -247
  112. package/.agents/core/skills/ddd/TROUBLESHOOTING.md +0 -19
  113. package/.agents/core/skills/ddd/VALIDATION.json +0 -12
  114. package/.agents/core/skills/ddd/ddd.md +0 -178
  115. package/.agents/core/skills/ddd/skill.yaml +0 -10
  116. package/.agents/core/skills/decisions/EXAMPLES.md +0 -35
  117. package/.agents/core/skills/decisions/SKILL.md +0 -90
  118. package/.agents/core/skills/decisions/TROUBLESHOOTING.md +0 -13
  119. package/.agents/core/skills/decisions/VALIDATION.json +0 -12
  120. package/.agents/core/skills/decisions/skill.yaml +0 -10
  121. package/.agents/core/skills/docker/EXAMPLES.md +0 -56
  122. package/.agents/core/skills/docker/SKILL.md +0 -63
  123. package/.agents/core/skills/docker/TROUBLESHOOTING.md +0 -18
  124. package/.agents/core/skills/docker/VALIDATION.json +0 -11
  125. package/.agents/core/skills/docker/skill.yaml +0 -23
  126. package/.agents/core/skills/fastapi/EXAMPLES.md +0 -36
  127. package/.agents/core/skills/fastapi/SKILL.md +0 -148
  128. package/.agents/core/skills/fastapi/TROUBLESHOOTING.md +0 -19
  129. package/.agents/core/skills/fastapi/VALIDATION.json +0 -12
  130. package/.agents/core/skills/fastapi/fastapi.md +0 -112
  131. package/.agents/core/skills/fastapi/skill.yaml +0 -10
  132. package/.agents/core/skills/generators/EXAMPLES.md +0 -19
  133. package/.agents/core/skills/generators/SKILL.md +0 -112
  134. package/.agents/core/skills/generators/TROUBLESHOOTING.md +0 -7
  135. package/.agents/core/skills/generators/VALIDATION.json +0 -12
  136. package/.agents/core/skills/generators/skill.yaml +0 -10
  137. package/.agents/core/skills/generators/templates/API.md +0 -77
  138. package/.agents/core/skills/generators/templates/ARCHITECTURE.md +0 -70
  139. package/.agents/core/skills/generators/templates/DATABASE.md +0 -42
  140. package/.agents/core/skills/generators/templates/DECISION.md +0 -46
  141. package/.agents/core/skills/generators/templates/PRD.md +0 -67
  142. package/.agents/core/skills/generators/templates/PROJECT_GRAPH.md +0 -56
  143. package/.agents/core/skills/generators/templates/ROADMAP.md +0 -51
  144. package/.agents/core/skills/generators/templates/TASKS.md +0 -43
  145. package/.agents/core/skills/generators/templates/UI.md +0 -73
  146. package/.agents/core/skills/graphify/EXAMPLES.md +0 -73
  147. package/.agents/core/skills/graphify/SKILL.md +0 -130
  148. package/.agents/core/skills/graphify/VALIDATION.json +0 -12
  149. package/.agents/core/skills/graphify/skill.yaml +0 -13
  150. package/.agents/core/skills/impeccable-design/EXAMPLES.md +0 -26
  151. package/.agents/core/skills/impeccable-design/SKILL.md +0 -201
  152. package/.agents/core/skills/impeccable-design/TROUBLESHOOTING.md +0 -19
  153. package/.agents/core/skills/impeccable-design/VALIDATION.json +0 -12
  154. package/.agents/core/skills/impeccable-design/skill.yaml +0 -14
  155. package/.agents/core/skills/interview-me/SKILL.md +0 -97
  156. package/.agents/core/skills/interview-me/VALIDATION.json +0 -12
  157. package/.agents/core/skills/interview-me/skill.yaml +0 -8
  158. package/.agents/core/skills/microservices/EXAMPLES.md +0 -38
  159. package/.agents/core/skills/microservices/SKILL.md +0 -164
  160. package/.agents/core/skills/microservices/TROUBLESHOOTING.md +0 -19
  161. package/.agents/core/skills/microservices/VALIDATION.json +0 -12
  162. package/.agents/core/skills/microservices/microservices.md +0 -119
  163. package/.agents/core/skills/microservices/skill.yaml +0 -10
  164. package/.agents/core/skills/minimalist-design/SKILL.md +0 -113
  165. package/.agents/core/skills/minimalist-design/VALIDATION.json +0 -12
  166. package/.agents/core/skills/minimalist-design/skill.yaml +0 -8
  167. package/.agents/core/skills/nestjs/EXAMPLES.md +0 -40
  168. package/.agents/core/skills/nestjs/SKILL.md +0 -139
  169. package/.agents/core/skills/nestjs/TROUBLESHOOTING.md +0 -19
  170. package/.agents/core/skills/nestjs/VALIDATION.json +0 -12
  171. package/.agents/core/skills/nestjs/nestjs.md +0 -103
  172. package/.agents/core/skills/nestjs/skill.yaml +0 -10
  173. package/.agents/core/skills/nextjs/EXAMPLES.md +0 -40
  174. package/.agents/core/skills/nextjs/SKILL.md +0 -163
  175. package/.agents/core/skills/nextjs/TROUBLESHOOTING.md +0 -19
  176. package/.agents/core/skills/nextjs/VALIDATION.json +0 -12
  177. package/.agents/core/skills/nextjs/nextjs.md +0 -67
  178. package/.agents/core/skills/nextjs/skill.yaml +0 -10
  179. package/.agents/core/skills/node/EXAMPLES.md +0 -80
  180. package/.agents/core/skills/node/SKILL.md +0 -128
  181. package/.agents/core/skills/node/TROUBLESHOOTING.md +0 -19
  182. package/.agents/core/skills/node/VALIDATION.json +0 -12
  183. package/.agents/core/skills/node/node.md +0 -87
  184. package/.agents/core/skills/node/skill.yaml +0 -10
  185. package/.agents/core/skills/performance/EXAMPLES.md +0 -30
  186. package/.agents/core/skills/performance/SKILL.md +0 -75
  187. package/.agents/core/skills/performance/TROUBLESHOOTING.md +0 -19
  188. package/.agents/core/skills/performance/VALIDATION.json +0 -12
  189. package/.agents/core/skills/performance/performance.md +0 -52
  190. package/.agents/core/skills/performance/skill.yaml +0 -10
  191. package/.agents/core/skills/react/EXAMPLES.md +0 -79
  192. package/.agents/core/skills/react/SKILL.md +0 -132
  193. package/.agents/core/skills/react/TROUBLESHOOTING.md +0 -19
  194. package/.agents/core/skills/react/VALIDATION.json +0 -12
  195. package/.agents/core/skills/react/react.md +0 -93
  196. package/.agents/core/skills/react/skill.yaml +0 -10
  197. package/.agents/core/skills/react-best-practices/SKILL.md +0 -155
  198. package/.agents/core/skills/react-best-practices/VALIDATION.json +0 -12
  199. package/.agents/core/skills/react-best-practices/skill.yaml +0 -10
  200. package/.agents/core/skills/redesign-audit/SKILL.md +0 -117
  201. package/.agents/core/skills/redesign-audit/VALIDATION.json +0 -12
  202. package/.agents/core/skills/redesign-audit/skill.yaml +0 -8
  203. package/.agents/core/skills/soft-design/SKILL.md +0 -108
  204. package/.agents/core/skills/soft-design/VALIDATION.json +0 -12
  205. package/.agents/core/skills/soft-design/skill.yaml +0 -8
  206. package/.agents/core/skills/state-management/EXAMPLES.md +0 -56
  207. package/.agents/core/skills/state-management/SKILL.md +0 -48
  208. package/.agents/core/skills/state-management/TROUBLESHOOTING.md +0 -18
  209. package/.agents/core/skills/state-management/VALIDATION.json +0 -11
  210. package/.agents/core/skills/state-management/skill.yaml +0 -22
  211. package/.agents/core/skills/subagent-orchestrator/SKILL.md +0 -100
  212. package/.agents/core/skills/subagent-orchestrator/VALIDATION.json +0 -12
  213. package/.agents/core/skills/subagent-orchestrator/skill.yaml +0 -8
  214. package/.agents/core/skills/system-design/EXAMPLES.md +0 -75
  215. package/.agents/core/skills/system-design/SKILL.md +0 -419
  216. package/.agents/core/skills/system-design/TROUBLESHOOTING.md +0 -19
  217. package/.agents/core/skills/system-design/VALIDATION.json +0 -12
  218. package/.agents/core/skills/system-design/skill.yaml +0 -13
  219. package/.agents/core/skills/system-design/system-design.md +0 -112
  220. package/.agents/core/skills/testing/EXAMPLES.md +0 -71
  221. package/.agents/core/skills/testing/SKILL.md +0 -70
  222. package/.agents/core/skills/testing/TROUBLESHOOTING.md +0 -18
  223. package/.agents/core/skills/testing/VALIDATION.json +0 -11
  224. package/.agents/core/skills/testing/skill.yaml +0 -26
  225. package/.agents/core/skills/typescript/EXAMPLES.md +0 -64
  226. package/.agents/core/skills/typescript/SKILL.md +0 -112
  227. package/.agents/core/skills/typescript/TROUBLESHOOTING.md +0 -19
  228. package/.agents/core/skills/typescript/VALIDATION.json +0 -12
  229. package/.agents/core/skills/typescript/skill.yaml +0 -10
  230. package/.agents/core/skills/typescript/typescript.md +0 -71
  231. package/.agents/core/skills/ui-design/EXAMPLES.md +0 -21
  232. package/.agents/core/skills/ui-design/SKILL.md +0 -124
  233. package/.agents/core/skills/ui-design/TROUBLESHOOTING.md +0 -19
  234. package/.agents/core/skills/ui-design/VALIDATION.json +0 -12
  235. package/.agents/core/skills/ui-design/skill.yaml +0 -10
  236. package/.agents/core/skills/ui-design/ui.md +0 -88
  237. package/.agents/core/skills/ui-ux-pro/EXAMPLES.md +0 -62
  238. package/.agents/core/skills/ui-ux-pro/SKILL.md +0 -375
  239. package/.agents/core/skills/ui-ux-pro/TROUBLESHOOTING.md +0 -19
  240. package/.agents/core/skills/ui-ux-pro/VALIDATION.json +0 -12
  241. package/.agents/core/skills/ui-ux-pro/skill.yaml +0 -13
  242. package/.agents/core/skills/ux-design/EXAMPLES.md +0 -36
  243. package/.agents/core/skills/ux-design/SKILL.md +0 -116
  244. package/.agents/core/skills/ux-design/TROUBLESHOOTING.md +0 -19
  245. package/.agents/core/skills/ux-design/VALIDATION.json +0 -12
  246. package/.agents/core/skills/ux-design/skill.yaml +0 -10
  247. package/.agents/core/skills/ux-design/ux.md +0 -80
  248. package/.agents/core/skills/vercel-optimize/SKILL.md +0 -83
  249. package/.agents/core/skills/vercel-optimize/VALIDATION.json +0 -12
  250. package/.agents/core/skills/vercel-optimize/skill.yaml +0 -10
  251. package/.agents/core/skills/web-accessibility/EXAMPLES.md +0 -39
  252. package/.agents/core/skills/web-accessibility/SKILL.md +0 -170
  253. package/.agents/core/skills/web-accessibility/TROUBLESHOOTING.md +0 -19
  254. package/.agents/core/skills/web-accessibility/VALIDATION.json +0 -12
  255. package/.agents/core/skills/web-accessibility/accessibility.md +0 -63
  256. package/.agents/core/skills/web-accessibility/skill.yaml +0 -10
  257. package/.agents/generated/claude/skills/adapters/SKILL.md +0 -126
  258. package/.agents/generated/claude/skills/architecture-diagrams/SKILL.md +0 -101
  259. package/.agents/generated/claude/skills/brutalist-design/SKILL.md +0 -145
  260. package/.agents/generated/claude/skills/database/SKILL.md +0 -191
  261. package/.agents/generated/claude/skills/ddd/SKILL.md +0 -305
  262. package/.agents/generated/claude/skills/decisions/SKILL.md +0 -134
  263. package/.agents/generated/claude/skills/docker/SKILL.md +0 -135
  264. package/.agents/generated/claude/skills/fastapi/SKILL.md +0 -200
  265. package/.agents/generated/claude/skills/generators/SKILL.md +0 -133
  266. package/.agents/generated/claude/skills/graphify/SKILL.md +0 -198
  267. package/.agents/generated/claude/skills/impeccable-design/SKILL.md +0 -241
  268. package/.agents/generated/claude/skills/interview-me/SKILL.md +0 -90
  269. package/.agents/generated/claude/skills/microservices/SKILL.md +0 -218
  270. package/.agents/generated/claude/skills/minimalist-design/SKILL.md +0 -108
  271. package/.agents/generated/claude/skills/nestjs/SKILL.md +0 -195
  272. package/.agents/generated/claude/skills/nextjs/SKILL.md +0 -219
  273. package/.agents/generated/claude/skills/node/SKILL.md +0 -224
  274. package/.agents/generated/claude/skills/performance/SKILL.md +0 -121
  275. package/.agents/generated/claude/skills/react/SKILL.md +0 -227
  276. package/.agents/generated/claude/skills/react-best-practices/SKILL.md +0 -146
  277. package/.agents/generated/claude/skills/redesign-audit/SKILL.md +0 -112
  278. package/.agents/generated/claude/skills/soft-design/SKILL.md +0 -103
  279. package/.agents/generated/claude/skills/state-management/SKILL.md +0 -120
  280. package/.agents/generated/claude/skills/subagent-orchestrator/SKILL.md +0 -93
  281. package/.agents/generated/claude/skills/system-design/SKILL.md +0 -507
  282. package/.agents/generated/claude/skills/testing/SKILL.md +0 -157
  283. package/.agents/generated/claude/skills/typescript/SKILL.md +0 -192
  284. package/.agents/generated/claude/skills/ui-design/SKILL.md +0 -161
  285. package/.agents/generated/claude/skills/ui-ux-pro/SKILL.md +0 -451
  286. package/.agents/generated/claude/skills/ux-design/SKILL.md +0 -168
  287. package/.agents/generated/claude/skills/vercel-optimize/SKILL.md +0 -76
  288. package/.agents/generated/claude/skills/web-accessibility/SKILL.md +0 -225
  289. package/.agents/generated/gemini/skills/adapters/SKILL.md +0 -135
  290. package/.agents/generated/gemini/skills/architecture-diagrams/SKILL.md +0 -107
  291. package/.agents/generated/gemini/skills/brutalist-design/SKILL.md +0 -151
  292. package/.agents/generated/gemini/skills/database/SKILL.md +0 -200
  293. package/.agents/generated/gemini/skills/ddd/SKILL.md +0 -314
  294. package/.agents/generated/gemini/skills/decisions/SKILL.md +0 -143
  295. package/.agents/generated/gemini/skills/docker/SKILL.md +0 -144
  296. package/.agents/generated/gemini/skills/fastapi/SKILL.md +0 -209
  297. package/.agents/generated/gemini/skills/generators/SKILL.md +0 -142
  298. package/.agents/generated/gemini/skills/graphify/SKILL.md +0 -205
  299. package/.agents/generated/gemini/skills/impeccable-design/SKILL.md +0 -250
  300. package/.agents/generated/gemini/skills/interview-me/SKILL.md +0 -96
  301. package/.agents/generated/gemini/skills/microservices/SKILL.md +0 -227
  302. package/.agents/generated/gemini/skills/minimalist-design/SKILL.md +0 -114
  303. package/.agents/generated/gemini/skills/nestjs/SKILL.md +0 -204
  304. package/.agents/generated/gemini/skills/nextjs/SKILL.md +0 -298
  305. package/.agents/generated/gemini/skills/node/SKILL.md +0 -323
  306. package/.agents/generated/gemini/skills/performance/SKILL.md +0 -185
  307. package/.agents/generated/gemini/skills/react/SKILL.md +0 -332
  308. package/.agents/generated/gemini/skills/react-best-practices/SKILL.md +0 -152
  309. package/.agents/generated/gemini/skills/redesign-audit/SKILL.md +0 -118
  310. package/.agents/generated/gemini/skills/soft-design/SKILL.md +0 -109
  311. package/.agents/generated/gemini/skills/state-management/SKILL.md +0 -129
  312. package/.agents/generated/gemini/skills/subagent-orchestrator/SKILL.md +0 -99
  313. package/.agents/generated/gemini/skills/system-design/SKILL.md +0 -631
  314. package/.agents/generated/gemini/skills/testing/SKILL.md +0 -166
  315. package/.agents/generated/gemini/skills/typescript/SKILL.md +0 -275
  316. package/.agents/generated/gemini/skills/ui-design/SKILL.md +0 -170
  317. package/.agents/generated/gemini/skills/ui-ux-pro/SKILL.md +0 -460
  318. package/.agents/generated/gemini/skills/ux-design/SKILL.md +0 -177
  319. package/.agents/generated/gemini/skills/vercel-optimize/SKILL.md +0 -82
  320. package/.agents/generated/gemini/skills/web-accessibility/SKILL.md +0 -300
  321. package/.agents/mcp/runtime.py +0 -470
  322. package/.agents/mcp/server.mjs +0 -189271
  323. package/benchmarks/gemini-issues.js +0 -533
@@ -1,533 +0,0 @@
1
- #!/usr/bin/env node
2
-
3
- /*
4
- * A reproducible, paired benchmark for ContextOS skills.
5
- *
6
- * It discovers closed JavaScript GitHub Issues linked to merged pull requests,
7
- * checks out the parent of each merge commit, and gives the same task to Gemini
8
- * twice: once with a neutral instruction and once with ContextOS skills. The
9
- * coding loop is intentionally controller-driven: the model may read files and
10
- * return a unified diff, but it never receives shell access.
11
- */
12
-
13
- 'use strict';
14
-
15
- const crypto = require('crypto');
16
- const fs = require('fs');
17
- const https = require('https');
18
- const os = require('os');
19
- const path = require('path');
20
- const { spawn } = require('child_process');
21
-
22
- // The benchmark belongs to the package but operates on the project from which
23
- // it is invoked, including that project's installed .agents directory.
24
- const ROOT = process.cwd();
25
- const DEFAULT_COUNT = 20;
26
- const DEFAULT_MAX_ITERATIONS = 6;
27
- const DEFAULT_MODEL = 'gemini-2.5-flash';
28
- const DEFAULT_QUERY = 'is:issue is:closed language:JavaScript label:bug';
29
- const MAX_FILE_BYTES = 24 * 1024;
30
- const MAX_OUTPUT_BYTES = 16 * 1024;
31
- const DEFAULT_SKILL_FILES = [
32
- '.agents/AGENTS.md',
33
- '.agents/core/skills/engineering-workflow/SKILL.md',
34
- '.agents/core/skills/system-design/SKILL.md',
35
- '.agents/core/skills/node/SKILL.md',
36
- '.agents/core/skills/security/SKILL.md',
37
- '.agents/core/skills/typescript/SKILL.md',
38
- ];
39
-
40
- function parseArgs(argv) {
41
- const options = {
42
- count: DEFAULT_COUNT,
43
- maxIterations: DEFAULT_MAX_ITERATIONS,
44
- model: DEFAULT_MODEL,
45
- query: DEFAULT_QUERY,
46
- output: path.join(ROOT, 'benchmarks', 'results'),
47
- tasks: null,
48
- allowCommands: false,
49
- dryRun: false,
50
- };
51
-
52
- for (let index = 0; index < argv.length; index++) {
53
- const arg = argv[index];
54
- const next = () => {
55
- const value = argv[++index];
56
- if (!value || value.startsWith('--')) throw new Error(`${arg} requires a value`);
57
- return value;
58
- };
59
- if (arg === '--count') options.count = Number(next());
60
- else if (arg === '--max-iterations') options.maxIterations = Number(next());
61
- else if (arg === '--model') options.model = next();
62
- else if (arg === '--query') options.query = next();
63
- else if (arg === '--output') options.output = path.resolve(next());
64
- else if (arg === '--tasks') options.tasks = path.resolve(next());
65
- else if (arg === '--allow-commands') options.allowCommands = true;
66
- else if (arg === '--dry-run') options.dryRun = true;
67
- else if (arg === '--help' || arg === '-h') options.help = true;
68
- else throw new Error(`Unknown option: ${arg}`);
69
- }
70
-
71
- if (!Number.isInteger(options.count) || options.count < 1 || options.count > 100) {
72
- throw new Error('--count must be an integer from 1 to 100');
73
- }
74
- if (!Number.isInteger(options.maxIterations) || options.maxIterations < 1 || options.maxIterations > 12) {
75
- throw new Error('--max-iterations must be an integer from 1 to 12');
76
- }
77
- return options;
78
- }
79
-
80
- function printHelp() {
81
- console.log(`ContextOS Gemini Issue Benchmark
82
-
83
- Usage:
84
- npm run benchmark -- [options]
85
-
86
- Options:
87
- --count <n> Issues to discover (default: ${DEFAULT_COUNT})
88
- --tasks <file> Re-run a saved, reproducible task manifest
89
- --query <github-query> GitHub issue search query
90
- --model <name> Gemini model (default: ${DEFAULT_MODEL})
91
- --max-iterations <n> Maximum model turns per run (default: ${DEFAULT_MAX_ITERATIONS})
92
- --output <directory> Report directory (default: benchmarks/results)
93
- --allow-commands Required before cloning repos or running setup/tests
94
- --dry-run Discover and validate tasks, but do not call Gemini or clone
95
-
96
- Environment:
97
- GEMINI_API_KEY Required for benchmark runs
98
- GITHUB_TOKEN Strongly recommended; discovery of 20 linked issues exceeds anonymous API limits
99
-
100
- The benchmark writes an immutable task manifest containing issue URLs and base
101
- commits before it executes. Re-run that manifest with --tasks for comparable results.`);
102
- }
103
-
104
- function truncate(value, limit = MAX_OUTPUT_BYTES) {
105
- const text = String(value || '');
106
- return text.length <= limit ? text : `${text.slice(0, limit)}\n…[truncated]`;
107
- }
108
-
109
- function safeRelativePath(repository, requestedPath) {
110
- if (typeof requestedPath !== 'string' || requestedPath.length === 0 || requestedPath.length > 240) return null;
111
- const resolved = path.resolve(repository, requestedPath);
112
- const relative = path.relative(repository, resolved);
113
- if (relative === '' || relative.startsWith('..') || path.isAbsolute(relative)) return null;
114
- return resolved;
115
- }
116
-
117
- function readRequestedFiles(repository, requestedPaths) {
118
- const files = {};
119
- for (const requestedPath of (requestedPaths || []).slice(0, 8)) {
120
- const resolved = safeRelativePath(repository, requestedPath);
121
- if (!resolved || !fs.existsSync(resolved) || !fs.statSync(resolved).isFile()) continue;
122
- if (fs.statSync(resolved).size > MAX_FILE_BYTES) continue;
123
- files[requestedPath] = fs.readFileSync(resolved, 'utf8');
124
- }
125
- return files;
126
- }
127
-
128
- function parseJsonObject(text) {
129
- const clean = String(text || '').trim().replace(/^```(?:json)?\s*/i, '').replace(/\s*```$/, '');
130
- const first = clean.indexOf('{');
131
- const last = clean.lastIndexOf('}');
132
- if (first < 0 || last <= first) throw new Error('Model response did not contain a JSON object');
133
- return JSON.parse(clean.slice(first, last + 1));
134
- }
135
-
136
- function parseAgentReply(text) {
137
- const reply = parseJsonObject(text);
138
- if (!Array.isArray(reply.read)) reply.read = [];
139
- if (typeof reply.patch !== 'string') reply.patch = '';
140
- if (typeof reply.ready !== 'boolean') reply.ready = false;
141
- if (typeof reply.summary !== 'string') reply.summary = '';
142
- if (reply.patch && !reply.patch.startsWith('diff --git ')) {
143
- throw new Error('Model patch must be a unified git diff starting with "diff --git"');
144
- }
145
- return reply;
146
- }
147
-
148
- function run(command, args, options = {}) {
149
- const { cwd, timeoutMs = 120_000, allowFailure = false } = options;
150
- return new Promise((resolve, reject) => {
151
- const child = spawn(command, args, { cwd, shell: false, windowsHide: true });
152
- let stdout = '';
153
- let stderr = '';
154
- let timedOut = false;
155
- const timer = setTimeout(() => {
156
- timedOut = true;
157
- child.kill('SIGTERM');
158
- }, timeoutMs);
159
- const append = (current, chunk) => truncate(current + chunk.toString(), MAX_OUTPUT_BYTES);
160
- child.stdout.on('data', chunk => { stdout = append(stdout, chunk); });
161
- child.stderr.on('data', chunk => { stderr = append(stderr, chunk); });
162
- child.on('error', error => {
163
- clearTimeout(timer);
164
- reject(error);
165
- });
166
- child.on('close', code => {
167
- clearTimeout(timer);
168
- const result = { code, stdout, stderr, timedOut, passed: code === 0 && !timedOut };
169
- if (!result.passed && !allowFailure) {
170
- const error = new Error(`${command} exited with code ${code}${timedOut ? ' (timed out)' : ''}`);
171
- error.result = result;
172
- reject(error);
173
- } else {
174
- resolve(result);
175
- }
176
- });
177
- });
178
- }
179
-
180
- function requestJson(url, { method = 'GET', headers = {}, body = null } = {}) {
181
- return new Promise((resolve, reject) => {
182
- const request = https.request(url, { method, headers }, response => {
183
- let raw = '';
184
- response.setEncoding('utf8');
185
- response.on('data', chunk => { raw = truncate(raw + chunk, MAX_OUTPUT_BYTES); });
186
- response.on('end', () => {
187
- try {
188
- resolve({ status: response.statusCode, body: JSON.parse(raw) });
189
- } catch {
190
- resolve({ status: response.statusCode, body: { message: raw } });
191
- }
192
- });
193
- });
194
- request.setTimeout(30_000, () => request.destroy(new Error('HTTP request timed out')));
195
- request.on('error', reject);
196
- if (body) request.write(body);
197
- request.end();
198
- });
199
- }
200
-
201
- async function githubJson(endpoint, token) {
202
- const response = await requestJson(`https://api.github.com${endpoint}`, {
203
- headers: {
204
- Accept: 'application/vnd.github+json',
205
- 'X-GitHub-Api-Version': '2022-11-28',
206
- 'User-Agent': 'contextos-agents-benchmark',
207
- ...(token ? { Authorization: `Bearer ${token}` } : {}),
208
- },
209
- });
210
- if (response.status < 200 || response.status >= 300) {
211
- throw new Error(`GitHub API ${response.status}: ${truncate(JSON.stringify(response.body), 500)}`);
212
- }
213
- return response.body;
214
- }
215
-
216
- function parseRepositoryFromApiUrl(repositoryUrl) {
217
- const match = String(repositoryUrl).match(/\/repos\/([^/]+)\/([^/]+)$/);
218
- if (!match) throw new Error(`Unexpected GitHub repository URL: ${repositoryUrl}`);
219
- return { owner: match[1], repo: match[2] };
220
- }
221
-
222
- async function resolveIssueTask(issue, token) {
223
- const { owner, repo } = parseRepositoryFromApiUrl(issue.repository_url);
224
- const prefix = `/repos/${encodeURIComponent(owner)}/${encodeURIComponent(repo)}`;
225
- const timeline = await githubJson(`${prefix}/issues/${issue.number}/timeline`, token);
226
- const source = timeline.find(event => event.event === 'cross-referenced' && event.source?.issue?.pull_request)?.source?.issue;
227
- if (!source?.number) return null;
228
-
229
- const pull = await githubJson(`${prefix}/pulls/${source.number}`, token);
230
- if (!pull.merged_at || !pull.merge_commit_sha) return null;
231
- const mergeCommit = await githubJson(`${prefix}/commits/${pull.merge_commit_sha}`, token);
232
- const baseSha = mergeCommit.parents?.[0]?.sha;
233
- if (!baseSha) return null;
234
-
235
- let packageJson;
236
- try {
237
- const content = await githubJson(`${prefix}/contents/package.json?ref=${encodeURIComponent(baseSha)}`, token);
238
- packageJson = JSON.parse(Buffer.from(content.content, 'base64').toString('utf8'));
239
- } catch {
240
- return null;
241
- }
242
- if (!packageJson.scripts?.test) return null;
243
-
244
- return {
245
- id: `${owner}/${repo}#${issue.number}`,
246
- repository: `${owner}/${repo}`,
247
- issueNumber: issue.number,
248
- issueUrl: issue.html_url,
249
- title: issue.title,
250
- body: issue.body || '',
251
- baseSha,
252
- linkedPullRequest: source.html_url,
253
- setup: ['npm', 'ci', '--ignore-scripts'],
254
- test: ['npm', 'test'],
255
- };
256
- }
257
-
258
- async function discoverTasks(count, query, token) {
259
- if (!token && count > 5) {
260
- throw new Error('Set GITHUB_TOKEN to discover 20 tasks without GitHub API rate-limit failures.');
261
- }
262
- const tasks = [];
263
- let page = 1;
264
- while (tasks.length < count && page <= 10) {
265
- const search = await githubJson(`/search/issues?q=${encodeURIComponent(query)}&per_page=100&page=${page}`, token);
266
- const items = search.items || [];
267
- if (items.length === 0) break;
268
- for (const issue of items) {
269
- if (tasks.length === count) break;
270
- if (issue.pull_request) continue;
271
- try {
272
- const task = await resolveIssueTask(issue, token);
273
- if (task) tasks.push(task);
274
- } catch (error) {
275
- console.warn(`[skip] ${issue.html_url}: ${error.message}`);
276
- }
277
- }
278
- page++;
279
- }
280
- if (tasks.length < count) {
281
- throw new Error(`Only resolved ${tasks.length}/${count} reproducible issues. Broaden --query and try again.`);
282
- }
283
- return tasks;
284
- }
285
-
286
- function writeJson(file, value) {
287
- fs.mkdirSync(path.dirname(file), { recursive: true });
288
- fs.writeFileSync(file, `${JSON.stringify(value, null, 2)}\n`);
289
- }
290
-
291
- function readTaskManifest(file) {
292
- const parsed = JSON.parse(fs.readFileSync(file, 'utf8'));
293
- const tasks = Array.isArray(parsed) ? parsed : parsed.tasks;
294
- if (!Array.isArray(tasks) || tasks.length === 0) throw new Error('Task manifest must contain a non-empty tasks array');
295
- for (const task of tasks) {
296
- if (!/^[\w.-]+\/[\w.-]+$/.test(task.repository || '') || !/^[a-f0-9]{40}$/i.test(task.baseSha || '')) {
297
- throw new Error(`Invalid repository or baseSha in task: ${task.id || task.issueUrl || 'unknown'}`);
298
- }
299
- if (!Array.isArray(task.setup) || !Array.isArray(task.test) || !task.issueUrl) {
300
- throw new Error(`Task ${task.id || task.issueUrl} lacks setup, test, or issueUrl`);
301
- }
302
- }
303
- return tasks;
304
- }
305
-
306
- function loadSkillContext() {
307
- return DEFAULT_SKILL_FILES
308
- .filter(file => fs.existsSync(path.join(ROOT, file)))
309
- .map(file => `\n===== ${file} =====\n${fs.readFileSync(path.join(ROOT, file), 'utf8')}`)
310
- .join('\n');
311
- }
312
-
313
- function buildAgentPrompt(task, mode, initialFiles, feedback = '') {
314
- const skills = mode === 'with_skills'
315
- ? `\nContextOS skills are authoritative guidance for this run:\n${loadSkillContext()}`
316
- : '';
317
- return `You are repairing a real JavaScript repository. You have no shell or network access.\n\nIssue: ${task.title}\n${task.body}\nIssue URL: ${task.issueUrl}\n\nReturn exactly one JSON object, with no markdown fence:\n{"read":["relative/file.js"],"patch":"diff --git ... or empty string","ready":false,"summary":"short explanation"}\n\nRules:\n- Request at most 8 repository files per turn before editing.\n- Only return a valid unified git diff in patch; do not invent command output.\n- Do not change lockfiles, CI, generated files, or dependencies unless the issue requires it.\n- Set ready to true only after you have supplied all required changes.\n- The controller will apply your patch and return test output.\n\nInitial files:\n${JSON.stringify(initialFiles, null, 2)}${feedback ? `\n\nController feedback:\n${feedback}` : ''}${skills}`;
318
- }
319
-
320
- function extractText(payload) {
321
- if (typeof payload?.output_text === 'string') return payload.output_text;
322
- const chunks = [];
323
- const visit = value => {
324
- if (Array.isArray(value)) return value.forEach(visit);
325
- if (!value || typeof value !== 'object') return;
326
- if (typeof value.text === 'string' && (!value.type || /text/.test(value.type))) {
327
- chunks.push(value.text);
328
- }
329
- for (const key of ['output', 'outputs', 'content', 'parts', 'candidates', 'message', 'delta']) {
330
- if (value[key]) visit(value[key]);
331
- }
332
- };
333
- visit(payload.output || payload.outputs || payload.candidates);
334
- return chunks.join('\n');
335
- }
336
-
337
- async function callGemini(apiKey, model, input, systemInstruction) {
338
- const started = Date.now();
339
- const response = await requestJson('https://generativelanguage.googleapis.com/v1beta/interactions', {
340
- method: 'POST',
341
- headers: { 'Content-Type': 'application/json', 'x-goog-api-key': apiKey },
342
- body: JSON.stringify({
343
- model,
344
- input,
345
- system_instruction: systemInstruction,
346
- generation_config: { temperature: 0 },
347
- }),
348
- });
349
- const payload = response.body;
350
- if (response.status < 200 || response.status >= 300) throw new Error(`Gemini API ${response.status}: ${truncate(JSON.stringify(payload), 800)}`);
351
- const text = extractText(payload);
352
- if (!text) throw new Error('Gemini returned no text output');
353
- return { text, latencyMs: Date.now() - started, usage: payload.usage_metadata || payload.usageMetadata || null };
354
- }
355
-
356
- async function gitOutput(repository, args) {
357
- const result = await run('git', args, { cwd: repository, allowFailure: true });
358
- return result.stdout;
359
- }
360
-
361
- async function cloneTask(task, workspace) {
362
- const destination = path.join(workspace, task.id.replace(/[^a-zA-Z0-9._-]/g, '-'));
363
- await run('git', ['clone', '--filter=blob:none', '--no-checkout', `https://github.com/${task.repository}.git`, destination], { cwd: workspace, timeoutMs: 300_000 });
364
- await run('git', ['checkout', '--detach', task.baseSha], { cwd: destination, timeoutMs: 120_000 });
365
- return destination;
366
- }
367
-
368
- async function applyPatch(repository, patch) {
369
- const patchFile = path.join(os.tmpdir(), `contextos-benchmark-${crypto.randomUUID()}.patch`);
370
- fs.writeFileSync(patchFile, patch);
371
- try {
372
- return await run('git', ['apply', '--whitespace=nowarn', patchFile], { cwd: repository, allowFailure: true });
373
- } finally {
374
- fs.rmSync(patchFile, { force: true });
375
- }
376
- }
377
-
378
- async function runTaskCommand(repository, command, timeoutMs = 300_000) {
379
- if (!Array.isArray(command) || command.length === 0) throw new Error('Task command must be a non-empty argument array');
380
- return run(command[0], command.slice(1), { cwd: repository, timeoutMs, allowFailure: true });
381
- }
382
-
383
- async function runAgent(task, mode, options, workspace) {
384
- const repository = await cloneTask(task, workspace);
385
- const setup = await runTaskCommand(repository, task.setup);
386
- if (!setup.passed) return { task, mode, status: 'setup_failed', setup, iterations: 0, score: 0 };
387
-
388
- const initialFiles = readRequestedFiles(repository, ['package.json', 'README.md']);
389
- let feedback = '';
390
- let lastTest = null;
391
- let lastReply = null;
392
- const turns = [];
393
- for (let iteration = 1; iteration <= options.maxIterations; iteration++) {
394
- const response = await callGemini(
395
- options.apiKey,
396
- options.model,
397
- buildAgentPrompt(task, mode, initialFiles, feedback),
398
- 'You are a careful software engineer. Follow the controller protocol exactly.'
399
- );
400
- let reply;
401
- try {
402
- reply = parseAgentReply(response.text);
403
- } catch (error) {
404
- return { task, mode, status: 'invalid_model_response', error: error.message, iterations: iteration, turns, score: 0 };
405
- }
406
- lastReply = reply;
407
- const files = readRequestedFiles(repository, reply.read);
408
- const turn = { iteration, latencyMs: response.latencyMs, usage: response.usage, requestedFiles: Object.keys(files), summary: reply.summary };
409
-
410
- if (reply.patch) {
411
- const applied = await applyPatch(repository, reply.patch);
412
- turn.patchApplied = applied.passed;
413
- if (!applied.passed) {
414
- feedback = `Patch was rejected:\n${truncate(applied.stderr || applied.stdout)}`;
415
- } else {
416
- lastTest = await runTaskCommand(repository, task.test);
417
- turn.test = lastTest;
418
- feedback = `Patch applied. Test result (exit ${lastTest.code}):\n${truncate(`${lastTest.stdout}\n${lastTest.stderr}`)}`;
419
- }
420
- } else if (Object.keys(files).length > 0) {
421
- feedback = `Requested file contents:\n${JSON.stringify(files, null, 2)}`;
422
- } else {
423
- feedback = 'No patch or readable file request was supplied. Return a valid next action.';
424
- }
425
- turns.push(turn);
426
-
427
- const diff = await gitOutput(repository, ['diff', '--no-ext-diff', '--unified=3']);
428
- if (reply.ready && lastTest?.passed && diff.trim()) break;
429
- }
430
-
431
- const diff = await gitOutput(repository, ['diff', '--no-ext-diff', '--unified=3']);
432
- const filesChanged = (await gitOutput(repository, ['diff', '--name-only'])).trim().split('\n').filter(Boolean);
433
- const ready = Boolean(lastReply?.ready && lastTest?.passed && diff.trim());
434
- const judge = await judgeChange(task, diff, lastTest, options);
435
- const score = scoreRun({ ready, test: lastTest, filesChanged, judge, iterations: turns.length });
436
- return { task, mode, status: ready ? 'ready' : 'not_ready', setup, test: lastTest, iterations: turns.length, turns, filesChanged, diff, judge, score };
437
- }
438
-
439
- async function judgeChange(task, diff, test, options) {
440
- if (!diff.trim()) return { score: 0, reason: 'No source change was produced.' };
441
- const prompt = `Evaluate a proposed fix for this GitHub Issue. Score only issue fidelity and code quality from 0 to 30. Do not reward tests merely passing; explain missing behavior. Return exactly JSON: {"score":number,"reason":"short"}.\n\nIssue: ${task.title}\n${task.body}\n\nDiff:\n${truncate(diff, 18_000)}\n\nTest output:\n${truncate(`${test?.stdout || ''}\n${test?.stderr || ''}`, 4_000)}`;
442
- try {
443
- const response = await callGemini(options.apiKey, options.model, prompt, 'You are an impartial senior code reviewer. Return JSON only.');
444
- const result = parseJsonObject(response.text);
445
- return { score: Math.max(0, Math.min(30, Number(result.score) || 0)), reason: String(result.reason || ''), latencyMs: response.latencyMs };
446
- } catch (error) {
447
- return { score: 0, reason: `Judge unavailable: ${error.message}` };
448
- }
449
- }
450
-
451
- function scoreRun({ ready, test, filesChanged, judge, iterations }) {
452
- const testScore = test?.passed ? 45 : 0;
453
- const scopeScore = filesChanged.length > 0 && filesChanged.length <= 6 ? 15 : filesChanged.length <= 12 ? 8 : 0;
454
- const safetyScore = filesChanged.every(file => !/(^|\/)(\.env|node_modules|\.github\/workflows)\b|(?:package-lock|yarn\.lock|pnpm-lock)/.test(file)) ? 10 : 0;
455
- const iterationScore = ready ? Math.max(0, 10 - Math.max(0, iterations - 1) * 2) : 0;
456
- return { total: testScore + scopeScore + safetyScore + iterationScore + (judge?.score || 0), testScore, scopeScore, safetyScore, iterationScore, judgeScore: judge?.score || 0 };
457
- }
458
-
459
- function average(values) {
460
- return values.length ? values.reduce((sum, value) => sum + value, 0) / values.length : 0;
461
- }
462
-
463
- function summarize(results) {
464
- const byMode = {};
465
- for (const mode of ['without_skills', 'with_skills']) {
466
- const runs = results.filter(result => result.mode === mode);
467
- byMode[mode] = {
468
- runs: runs.length,
469
- readyRate: runs.length ? runs.filter(run => run.status === 'ready').length / runs.length : 0,
470
- testPassRate: runs.length ? runs.filter(run => run.test?.passed).length / runs.length : 0,
471
- averageScore: average(runs.map(run => run.score?.total || 0)),
472
- averageIterations: average(runs.map(run => run.iterations || 0)),
473
- };
474
- }
475
- const paired = results.reduce((map, result) => {
476
- const entry = map.get(result.task.id) || {};
477
- entry[result.mode] = result;
478
- map.set(result.task.id, entry);
479
- return map;
480
- }, new Map());
481
- const deltas = [...paired.values()]
482
- .filter(pair => pair.without_skills && pair.with_skills)
483
- .map(pair => pair.with_skills.score.total - pair.without_skills.score.total);
484
- return { byMode, pairedTasks: deltas.length, meanSkillScoreDelta: average(deltas) };
485
- }
486
-
487
- function markdownReport(report) {
488
- const row = (name, values) => `| ${name} | ${values.runs} | ${(values.readyRate * 100).toFixed(1)}% | ${(values.testPassRate * 100).toFixed(1)}% | ${values.averageScore.toFixed(1)} | ${values.averageIterations.toFixed(2)} |`;
489
- return `# ContextOS Skills Benchmark\n\nModel: \`${report.model}\` \nTasks: ${report.tasks.length} \nPaired tasks: ${report.summary.pairedTasks}\n\n| Mode | Runs | Ready | Tests pass | Mean quality (0–110) | Mean turns |\n| --- | ---: | ---: | ---: | ---: | ---: |\n${row('Without skills', report.summary.byMode.without_skills)}\n${row('With ContextOS skills', report.summary.byMode.with_skills)}\n\nMean paired score delta (with skills − without): **${report.summary.meanSkillScoreDelta.toFixed(1)}**\n\n## Per task\n\n| Issue | Mode | Status | Turns | Score |\n| --- | --- | --- | ---: | ---: |\n${report.results.map(run => `| [${run.task.id}](${run.task.issueUrl}) | ${run.mode} | ${run.status} | ${run.iterations} | ${run.score?.total || 0} |`).join('\n')}\n`;
490
- }
491
-
492
- async function main() {
493
- const options = parseArgs(process.argv.slice(2));
494
- if (options.help) return printHelp();
495
- const token = process.env.GITHUB_TOKEN;
496
- const tasks = options.tasks ? readTaskManifest(options.tasks) : await discoverTasks(options.count, options.query, token);
497
- const manifestPath = options.tasks || path.join(options.output, `tasks-${new Date().toISOString().replace(/[:.]/g, '-')}.json`);
498
- if (!options.tasks) writeJson(manifestPath, { generatedAt: new Date().toISOString(), query: options.query, tasks });
499
- console.log(`Prepared ${tasks.length} reproducible GitHub Issue tasks: ${manifestPath}`);
500
-
501
- if (options.dryRun || !options.allowCommands) {
502
- console.log(options.dryRun ? 'Dry run complete; no model calls or repositories were executed.' : 'Pass --allow-commands to clone task repositories and run their declared setup/tests.');
503
- return;
504
- }
505
- if (!process.env.GEMINI_API_KEY) throw new Error('GEMINI_API_KEY is required for a benchmark run');
506
-
507
- const workspace = fs.mkdtempSync(path.join(os.tmpdir(), 'contextos-benchmark-'));
508
- try {
509
- const results = [];
510
- for (const task of tasks) {
511
- for (const mode of ['without_skills', 'with_skills']) {
512
- console.log(`\n[${mode}] ${task.id}`);
513
- results.push(await runAgent(task, mode, { ...options, apiKey: process.env.GEMINI_API_KEY }, workspace));
514
- }
515
- }
516
- const report = { generatedAt: new Date().toISOString(), model: options.model, tasks, results, summary: summarize(results) };
517
- const stamp = new Date().toISOString().replace(/[:.]/g, '-');
518
- writeJson(path.join(options.output, `report-${stamp}.json`), report);
519
- fs.writeFileSync(path.join(options.output, `report-${stamp}.md`), markdownReport(report));
520
- console.log(`\nBenchmark complete. Reports written to ${options.output}`);
521
- } finally {
522
- fs.rmSync(workspace, { recursive: true, force: true });
523
- }
524
- }
525
-
526
- module.exports = { buildAgentPrompt, parseAgentReply, parseArgs, scoreRun, summarize, safeRelativePath };
527
-
528
- if (require.main === module) {
529
- main().catch(error => {
530
- console.error(`[ERROR] ${error.message}`);
531
- process.exit(1);
532
- });
533
- }