cmpnd 0.7.1__tar.gz → 0.7.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (227) hide show
  1. {cmpnd-0.7.1 → cmpnd-0.7.2}/PKG-INFO +2 -1
  2. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/_gepa_patch.py +9 -2
  3. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/optimizers/gepa.py +6 -2
  4. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/optimizers/gepa_callback.py +1 -1
  5. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/optimizers/tracker.py +11 -4
  6. {cmpnd-0.7.1 → cmpnd-0.7.2}/pyproject.toml +38 -1
  7. {cmpnd-0.7.1 → cmpnd-0.7.2}/scripts/seed_review_run.py +290 -7
  8. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/committee_task.py +19 -4
  9. cmpnd-0.7.2/tests/stress/_provider.py +276 -0
  10. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/stress/test_deploy_programs.py +38 -8
  11. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/stress/test_optimize_loop.py +8 -1
  12. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/stress/test_optimize_loop_wasm.py +71 -24
  13. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/stress/test_optimize_stress.py +6 -1
  14. cmpnd-0.7.2/tests/stress/test_provider_roles.py +221 -0
  15. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/stress/test_real_provider_twins.py +19 -2
  16. cmpnd-0.7.2/tests/stress/test_rlm_wasm.py +411 -0
  17. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_clustering/conftest.py +2 -9
  18. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_clustering/test_edge_cases.py +2 -7
  19. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_clustering/test_hash_consistency.py +4 -11
  20. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_clustering/test_type_conversion.py +3 -8
  21. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_deploy.py +50 -0
  22. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_optimize_terminal_failure.py +78 -0
  23. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_seed_review_run.py +10 -4
  24. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_tracked_gepa.py +5 -1
  25. {cmpnd-0.7.1 → cmpnd-0.7.2}/uv.lock +58 -94
  26. cmpnd-0.7.1/tests/stress/_provider.py +0 -91
  27. {cmpnd-0.7.1 → cmpnd-0.7.2}/.gitignore +0 -0
  28. {cmpnd-0.7.1 → cmpnd-0.7.2}/CLAUDE.md +0 -0
  29. {cmpnd-0.7.1 → cmpnd-0.7.2}/README.md +0 -0
  30. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/__init__.py +0 -0
  31. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/_program_patch.py +0 -0
  32. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/_rlm_patch.py +0 -0
  33. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/callback.py +0 -0
  34. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/cli/__init__.py +0 -0
  35. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/cli/api_client.py +0 -0
  36. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/cli/commands/__init__.py +0 -0
  37. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/cli/commands/admin.py +0 -0
  38. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/cli/commands/auth.py +0 -0
  39. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/cli/commands/datasets.py +0 -0
  40. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/cli/commands/deployments.py +0 -0
  41. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/cli/commands/evals.py +0 -0
  42. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/cli/commands/login.py +0 -0
  43. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/cli/commands/optimizations.py +0 -0
  44. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/cli/commands/org.py +0 -0
  45. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/cli/commands/programs.py +0 -0
  46. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/cli/commands/top.py +0 -0
  47. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/cli/commands/traces.py +0 -0
  48. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/cli/credentials.py +0 -0
  49. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/cli/shell.py +0 -0
  50. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/cli/trace_render.py +0 -0
  51. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/cli/verbs.py +0 -0
  52. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/configuration.py +0 -0
  53. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/context.py +0 -0
  54. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/dataset_sync.py +0 -0
  55. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/datasets.py +0 -0
  56. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/decorators.py +0 -0
  57. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/deployment.py +0 -0
  58. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/eval_handler.py +0 -0
  59. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/execution.py +0 -0
  60. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/exporter.py +0 -0
  61. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/exporter_http.py +0 -0
  62. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/helpers.py +0 -0
  63. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/identity.py +0 -0
  64. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/ir/__init__.py +0 -0
  65. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/ir/decode.py +0 -0
  66. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/ir/encode.py +0 -0
  67. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/ir/hash.py +0 -0
  68. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/ir/program.py +0 -0
  69. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/ir/reflection.py +0 -0
  70. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/ir/types.py +0 -0
  71. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/ir/xxh64.py +0 -0
  72. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/models.py +0 -0
  73. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/optimization.py +0 -0
  74. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/optimizers/__init__.py +0 -0
  75. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/optimizers/resume.py +0 -0
  76. {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/packaging.py +0 -0
  77. {cmpnd-0.7.1 → cmpnd-0.7.2}/docs/2026-04-02-code-review.md +0 -0
  78. {cmpnd-0.7.1 → cmpnd-0.7.2}/docs/plans/2026-04-05-train-val-dataset-capture-design.md +0 -0
  79. {cmpnd-0.7.1 → cmpnd-0.7.2}/docs/plans/2026-04-05-train-val-dataset-capture.md +0 -0
  80. {cmpnd-0.7.1 → cmpnd-0.7.2}/docs/sdk-instrumentation.md +0 -0
  81. {cmpnd-0.7.1 → cmpnd-0.7.2}/examples/local_ollama.py +0 -0
  82. {cmpnd-0.7.1 → cmpnd-0.7.2}/scripts/capture_review_run.py +0 -0
  83. {cmpnd-0.7.1 → cmpnd-0.7.2}/scripts/fixtures/review_run.json.gz +0 -0
  84. {cmpnd-0.7.1 → cmpnd-0.7.2}/scripts/gepa_resume_bug.py +0 -0
  85. {cmpnd-0.7.1 → cmpnd-0.7.2}/stubs/dspy/__init__.pyi +0 -0
  86. {cmpnd-0.7.1 → cmpnd-0.7.2}/stubs/dspy/primitives/__init__.pyi +0 -0
  87. {cmpnd-0.7.1 → cmpnd-0.7.2}/stubs/dspy/primitives/prediction.pyi +0 -0
  88. {cmpnd-0.7.1 → cmpnd-0.7.2}/stubs/dspy/utils/__init__.pyi +0 -0
  89. {cmpnd-0.7.1 → cmpnd-0.7.2}/stubs/dspy/utils/callback.pyi +0 -0
  90. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/__init__.py +0 -0
  91. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/_exporter_helpers.py +0 -0
  92. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/_identity_helpers.py +0 -0
  93. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/__init__.py +0 -0
  94. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/conftest.py +0 -0
  95. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/data/committee_train.ndjson.gz +0 -0
  96. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/data/committee_val.ndjson.gz +0 -0
  97. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/fake_lm.py +0 -0
  98. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/helpers.py +0 -0
  99. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/test_capture_prompts.py +0 -0
  100. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/test_composite_module.py +0 -0
  101. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/test_dataset_auto_creation.py +0 -0
  102. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/test_deploy_execute.py +0 -0
  103. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/test_deploy_programs.py +0 -0
  104. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/test_edge_cases.py +0 -0
  105. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/test_example_hash.py +0 -0
  106. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/test_gepa_optimization.py +0 -0
  107. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/test_lm_accuracy.py +0 -0
  108. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/test_module_tracing.py +0 -0
  109. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/test_optimization_lineage.py +0 -0
  110. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/test_optimization_metric_identity.py +0 -0
  111. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/test_optimize_failures.py +0 -0
  112. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/test_optimize_loop.py +0 -0
  113. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/test_optimize_review_committee.py +0 -0
  114. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/test_optimize_server_owned_completion.py +0 -0
  115. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/test_parallel_export.py +0 -0
  116. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/test_project_attribution.py +0 -0
  117. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/test_react_tool_spans.py +0 -0
  118. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/test_save_load.py +0 -0
  119. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/test_trace_structure.py +0 -0
  120. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/test_user_attribution.py +0 -0
  121. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/identity_vectors/01_predict_qa.json +0 -0
  122. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/identity_vectors/02_predict_renamed_field.json +0 -0
  123. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/identity_vectors/03_chain_of_thought_qa.json +0 -0
  124. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/identity_vectors/04_predict_richer_types.json +0 -0
  125. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/identity_vectors/05_composite_two_predicts.json +0 -0
  126. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/identity_vectors/06_multichaincomparison.json +0 -0
  127. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/identity_vectors/07_program_of_thought.json +0 -0
  128. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/identity_vectors/08_opaque_module.json +0 -0
  129. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/identity_vectors/09_predict_pydantic_field.json +0 -0
  130. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/identity_vectors/10_react.json +0 -0
  131. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/identity_vectors/11_refine.json +0 -0
  132. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/identity_vectors/12_retrieve.json +0 -0
  133. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/identity_vectors/13_knn.json +0 -0
  134. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/identity_vectors/14_parallel.json +0 -0
  135. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/identity_vectors/README.md +0 -0
  136. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/integration/__init__.py +0 -0
  137. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/integration/conftest.py +0 -0
  138. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/integration/test_cli_deploy_smoketest.py +0 -0
  139. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/integration/test_logs_renders_traces.py +0 -0
  140. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/integration/test_optimizing_progress.py +0 -0
  141. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/integration/test_unhappy_path.py +0 -0
  142. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/parity/__init__.py +0 -0
  143. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/parity/conftest.py +0 -0
  144. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/parity/seed_parity.py +0 -0
  145. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/parity/test_auth.py +0 -0
  146. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/parity/test_datasets.py +0 -0
  147. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/parity/test_deployments.py +0 -0
  148. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/parity/test_evals.py +0 -0
  149. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/parity/test_health.py +0 -0
  150. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/parity/test_modules.py +0 -0
  151. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/parity/test_optimizations.py +0 -0
  152. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/parity/test_programs.py +0 -0
  153. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/parity/test_signatures.py +0 -0
  154. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/parity/test_trace_streaming.py +0 -0
  155. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/parity/test_traces.py +0 -0
  156. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/stress/__init__.py +0 -0
  157. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/stress/_reliability.py +0 -0
  158. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/stress/mint_bedrock_token.py +0 -0
  159. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/stress/test_multi_client_stress.py +0 -0
  160. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/stress/test_optimize_client_committee.py +0 -0
  161. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/stress/test_reliability.py +0 -0
  162. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/stress/test_reliability_classify.py +0 -0
  163. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_callback.py +0 -0
  164. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_cli/__init__.py +0 -0
  165. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_cli/conftest.py +0 -0
  166. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_cli/test_admin.py +0 -0
  167. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_cli/test_api_client.py +0 -0
  168. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_cli/test_auth.py +0 -0
  169. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_cli/test_credentials.py +0 -0
  170. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_cli/test_datasets.py +0 -0
  171. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_cli/test_deployments.py +0 -0
  172. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_cli/test_evals.py +0 -0
  173. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_cli/test_login.py +0 -0
  174. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_cli/test_main.py +0 -0
  175. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_cli/test_optimizations.py +0 -0
  176. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_cli/test_org.py +0 -0
  177. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_cli/test_parity.py +0 -0
  178. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_cli/test_programs.py +0 -0
  179. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_cli/test_shell.py +0 -0
  180. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_cli/test_top.py +0 -0
  181. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_cli/test_traces.py +0 -0
  182. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_clustering/__init__.py +0 -0
  183. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_cmp551_grouping_identity.py +0 -0
  184. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_configuration.py +0 -0
  185. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_context.py +0 -0
  186. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_context_propagation.py +0 -0
  187. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_dataset_polling.py +0 -0
  188. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_dataset_sync.py +0 -0
  189. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_datasets.py +0 -0
  190. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_decorators.py +0 -0
  191. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_eval.py +0 -0
  192. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_eval_start.py +0 -0
  193. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_exception_paths.py +0 -0
  194. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_execute.py +0 -0
  195. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_exporter.py +0 -0
  196. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_exporter_real.py +0 -0
  197. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_fake_lm.py +0 -0
  198. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_gepa_callback.py +0 -0
  199. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_gepa_integration.py +0 -0
  200. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_gepa_resume.py +0 -0
  201. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_identity.py +0 -0
  202. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_import_weight.py +0 -0
  203. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_integration.py +0 -0
  204. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_ir_encoder_properties.py +0 -0
  205. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_ir_golden_vectors.py +0 -0
  206. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_ir_roundtrip.py +0 -0
  207. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_models.py +0 -0
  208. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_optimization_progress.py +0 -0
  209. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_optimization_start.py +0 -0
  210. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_optimize.py +0 -0
  211. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_optimize_lift.py +0 -0
  212. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_packaging.py +0 -0
  213. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_packaging_callable_source.py +0 -0
  214. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_packaging_program_source.py +0 -0
  215. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_program_lineage_persist.py +0 -0
  216. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_project_payload.py +0 -0
  217. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_reflector_properties.py +0 -0
  218. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_reflector_smoke.py +0 -0
  219. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_repro_reactv2_bugs.py +0 -0
  220. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_rlm_patch.py +0 -0
  221. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_schemas.py +0 -0
  222. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_stats_export.py +0 -0
  223. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_trace_render.py +0 -0
  224. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_tracker_enrich.py +0 -0
  225. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_truncation.py +0 -0
  226. {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_xxh64.py +0 -0
  227. {cmpnd-0.7.1 → cmpnd-0.7.2}/todo.md +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: cmpnd
3
- Version: 0.7.1
3
+ Version: 0.7.2
4
4
  Summary: DSPy observability and deployment SDK for cmpnd
5
5
  Project-URL: Homepage, https://cmpnd.ai
6
6
  Author-email: cmpnd <hello@cmpnd.ai>
@@ -27,6 +27,7 @@ Requires-Dist: basedpyright>=1.39; extra == 'dev'
27
27
  Requires-Dist: boto3>=1.34; extra == 'dev'
28
28
  Requires-Dist: hypothesis>=6.100; extra == 'dev'
29
29
  Requires-Dist: jsonschema>=4.0; extra == 'dev'
30
+ Requires-Dist: numpy>=1.26; extra == 'dev'
30
31
  Requires-Dist: pytest-asyncio>=0.23; extra == 'dev'
31
32
  Requires-Dist: pytest-rerunfailures>=14.0; extra == 'dev'
32
33
  Requires-Dist: pytest-xdist>=3.0; extra == 'dev'
@@ -64,7 +64,7 @@ def patch_gepa() -> None:
64
64
  # This ensures each optimization gets its own run_id, even when
65
65
  # the same GEPA instance is reused (e.g., optimizing an already-optimized program).
66
66
  trk = tracker.OptimizationTracker(
67
- optimizer_type="GEPA",
67
+ optimizer_type="gepa",
68
68
  program_name="",
69
69
  )
70
70
  callback = gepa_callback.CmpndGEPACallback(trk, defer_finalize=True)
@@ -157,7 +157,14 @@ def patch_gepa() -> None:
157
157
 
158
158
  return result
159
159
 
160
- except Exception as e:
160
+ # `BaseException`, because the invariant does not get to have an exception: a
161
+ # run that ends by ANY route must transition the row start() sent, or the
162
+ # program page shows a perpetual in-progress optimization. A backend that
163
+ # rejected a model request raises a failure that deliberately does NOT derive
164
+ # from `Exception` — that is the only way past gepa's and dspy's own blanket
165
+ # `except Exception` clauses — so narrowing this one reopens exactly the bug
166
+ # this block exists for. Re-raised either way, so widening swallows nothing.
167
+ except BaseException as e:
161
168
  if callback is not None: # pyright: ignore[reportUnnecessaryComparison]
162
169
  # Flush a terminal 'failed', not just in-memory state — else the
163
170
  # running row start() sent lingers forever.
@@ -152,7 +152,7 @@ class TrackedGEPA:
152
152
 
153
153
  # Initialize optimization tracking
154
154
  self._tracker = tracker.OptimizationTracker(
155
- optimizer_type="GEPA",
155
+ optimizer_type="gepa",
156
156
  program_name=student.__class__.__name__,
157
157
  config=config,
158
158
  model_name=model_name,
@@ -218,7 +218,11 @@ class TrackedGEPA:
218
218
 
219
219
  return result
220
220
 
221
- except Exception as e:
221
+ # `BaseException` for the reason `_gepa_patch._patched_compile` documents: the
222
+ # row must transition however the run ends, and a rejected model request
223
+ # reaches here as something that is not an `Exception`, by design. Re-raised
224
+ # either way.
225
+ except BaseException as e:
222
226
  if self._tracker:
223
227
  # Flush a terminal 'failed', not just in-memory state — else the
224
228
  # running row start() sent lingers forever.
@@ -5,7 +5,7 @@ monkey-patching or wrapper classes. Replaces the monkey-patched
5
5
  InstrumentedDspyAdapter approach used by TrackedGEPA.
6
6
 
7
7
  Usage:
8
- tracker = OptimizationTracker(optimizer_type="GEPA", ...)
8
+ tracker = OptimizationTracker(optimizer_type="gepa", ...)
9
9
  callback = CmpndGEPACallback(tracker)
10
10
  gepa = dspy.GEPA(gepa_kwargs={"callbacks": [callback]})
11
11
  """
@@ -199,7 +199,7 @@ class OptimizationTracker:
199
199
 
200
200
  Usage:
201
201
  tracker = OptimizationTracker(
202
- optimizer_type="GEPA",
202
+ optimizer_type="gepa",
203
203
  program_name="MyProgram",
204
204
  config={"auto": "light", "seed": 42}
205
205
  )
@@ -634,14 +634,21 @@ class OptimizationTracker:
634
634
  except Exception as e:
635
635
  logger.warning(f"Failed to capture compiled program state: {e}")
636
636
 
637
- def set_error(self, exception: Exception) -> None:
638
- """Mark the optimization run as failed (in-memory only)."""
637
+ def set_error(self, exception: BaseException) -> None:
638
+ """Mark the optimization run as failed (in-memory only).
639
+
640
+ ``BaseException`` and not ``Exception``: a backend that rejected a model request
641
+ reports it as a failure that deliberately does not derive from ``Exception``,
642
+ since that is the only thing gepa's and dspy's blanket ``except Exception``
643
+ clauses do not swallow. A run that ended is a run that has to be recorded as
644
+ ended, whichever base class carried the reason.
645
+ """
639
646
  self.status = "failed"
640
647
  self.error_message = str(exception)
641
648
  self.end_time = datetime.now(timezone.utc)
642
649
  logger.error(f"Optimization failed: {exception}")
643
650
 
644
- def finalize_error(self, exception: Exception) -> None:
651
+ def finalize_error(self, exception: BaseException) -> None:
645
652
  """Mark the run failed AND flush the terminal 'failed' to the backend.
646
653
 
647
654
  The success path is finalize(); on an optimizer exception the run must
@@ -5,7 +5,7 @@ name = "cmpnd"
5
5
  # it's new) — no git tags. Bump with `uv version --bump {minor,patch}` or by
6
6
  # hand. Runtime `cmpnd.__version__` reads installed metadata, so it tracks
7
7
  # this automatically. See docs/plans/2026-07-02-sdk-pypi-tag-free-publish.md.
8
- version = "0.7.1"
8
+ version = "0.7.2"
9
9
  description = "DSPy observability and deployment SDK for cmpnd"
10
10
  readme = "README.md"
11
11
  requires-python = ">=3.10"
@@ -58,6 +58,10 @@ dev = [
58
58
  # litellm's bedrock/ provider imports boto3 in-process, so the client-side
59
59
  # Bedrock stress tests (committee twins) need it in the test env.
60
60
  "boto3>=1.34",
61
+ # test_reflector_smoke.py builds numpy fixtures. dspy <=3.2.x required numpy
62
+ # unconditionally so it rode in transitively; dspy 3.3.0 moved numpy behind a
63
+ # `numpy` extra, so declare it directly as the test dependency it always was.
64
+ "numpy>=1.26",
61
65
  ]
62
66
  # Deps for `deploy/deploy/stub/`, the local FastAPI server that emulates
63
67
  # the AWS deploy stack. The stub itself ships in the cmpnd-deploy
@@ -92,6 +96,39 @@ build-backend = "hatchling.build"
92
96
  [tool.hatch.build.targets.wheel]
93
97
  packages = ["cmpnd"]
94
98
 
99
+ [tool.pytest.ini_options]
100
+ # Deliberately minimal: `markers` and `--strict-markers`, nothing else. Setting
101
+ # `testpaths` or a broader `addopts` here would change how every existing
102
+ # invocation resolves (CI lanes, `uv run pytest tests/ -k "not e2e"`, a bare
103
+ # `pytest` from a subdirectory), and this file's job is only to make the marker
104
+ # vocabulary explicit.
105
+ #
106
+ # --strict-markers turns a typo'd marker into an error instead of a silent
107
+ # no-op. That matters most for `forall_workers`: it SELECTS the merge-queue's
108
+ # wasm lane, so a misspelling there would quietly select nothing and report a
109
+ # green lane that ran zero tests. `asyncio` and `flaky` are registered by
110
+ # pytest-asyncio / pytest-rerunfailures, so they are not listed here.
111
+ addopts = "--strict-markers"
112
+ markers = [
113
+ # Tests that must hold for EVERY worker backend — i.e. that provision,
114
+ # execute or optimize through a data plane, and so can distinguish the
115
+ # Lambda plane from the wasm plane from the on-prem worker. This is the
116
+ # selector for the per-plane stress lanes: the set worth running against a
117
+ # second worker is exactly the set claimed to hold for all of them.
118
+ #
119
+ # A stress test that deploys or executes gets this marker. Its absence on
120
+ # such a test is a bug; its PRESENCE on a test no worker can affect (one
121
+ # that only exercises client-side tracing, or serve's own health) is worse,
122
+ # because it costs a second real-provider run to prove nothing.
123
+ # See docs/plans/2026-07-29-wasm-stress-lane.md.
124
+ "forall_workers: must hold for every worker backend (reaches a data plane)",
125
+ # Pre-existing vocabulary, registered here because --strict-markers now
126
+ # requires it — not introduced by this block.
127
+ "parity: CLI dual-front-end parity (argparse vs. interactive shell)",
128
+ "e2e: needs a live platform; excluded from the fast lane via -k 'not e2e'",
129
+ "integration: crosses a component boundary (SDK ↔ serve ↔ worker)",
130
+ ]
131
+
95
132
  [tool.ruff]
96
133
  line-length = 120
97
134
  target-version = "py310"
@@ -33,12 +33,14 @@ from __future__ import annotations
33
33
 
34
34
  import datetime
35
35
  import gzip
36
+ import http.cookiejar
36
37
  import json
37
38
  import logging
38
39
  import os
39
40
  import pathlib
40
41
  import re
41
42
  import urllib.error
43
+ import urllib.parse
42
44
  import urllib.request
43
45
  import uuid
44
46
 
@@ -46,13 +48,19 @@ logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(mess
46
48
  log = logging.getLogger("seed_review_run")
47
49
 
48
50
  _FIXTURE = pathlib.Path(__file__).resolve().parent / "fixtures" / "review_run.json.gz"
49
- # Origin tag so seeded data is identifiable as such (never confused with real
50
- # customer runs). Overrides whatever the captured fixture carried. Each seeded
51
- # run also carries a `lifecycle-*` role tag (CMP-414) so its place in the
52
- # lifecycle demo is explicit — and so the failed run reads as intentional, not
53
- # an incident.
51
+ # Origin tag on ALL seeded rows so seeded data is identifiable as such (never
52
+ # confused with real customer runs). Overrides whatever the captured fixture
53
+ # carried. Each seeded run also carries a `lifecycle-*` role tag (CMP-414) so its
54
+ # place in the lifecycle demo is explicit — and so the failed run reads as
55
+ # intentional, not an incident.
54
56
  _SEED_TAGS = ["seed-data"]
55
- _TAGS_COMPLETE = _SEED_TAGS + ["lifecycle-complete"]
57
+ # `golden-path` marks ONLY the data that IS a DSPy-Meetup golden-path demo beat
58
+ # (CMP-634) — the completed committee run + its Predict traces, the ReAct
59
+ # committee trace, and the RLM notebook (completed + in-progress). It does NOT
60
+ # go on the lifecycle running/failed twins or the CMP-187 tool-demo traces, so
61
+ # `?tags=golden-path` surfaces exactly the demo tenant, nothing incidental.
62
+ _GOLDEN_PATH = "golden-path"
63
+ _TAGS_COMPLETE = _SEED_TAGS + [_GOLDEN_PATH, "lifecycle-complete"]
56
64
  _TAGS_RUNNING = _SEED_TAGS + ["lifecycle-running"]
57
65
  _TAGS_FAILED = _SEED_TAGS + ["lifecycle-failed"]
58
66
  _UUID_RE = re.compile(r"[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}")
@@ -248,6 +256,14 @@ _SOURCE_CATEGORIZER_FIXTURE = (
248
256
  _TOOL_KIND_DEMO_FIXTURE = (
249
257
  pathlib.Path(__file__).resolve().parents[2] / "serve" / "scripts" / "seed_fixtures" / "tool_kind_demo.json"
250
258
  )
259
+ # A dspy.ReAct(ExtractCommittee) run (email_body->committee) that reasons, calls
260
+ # a `lookup_disclaimer` tool to pull the "Paid for by" line, then extracts the
261
+ # committee (CMP-634 golden-path). Flat-replay format like the two fixtures
262
+ # above; program_name "ReAct" so the sidebar/Type badge reads ReAct. Loaded
263
+ # lazily so a missing file degrades to skipping that seed.
264
+ _REACT_COMMITTEE_FIXTURE = (
265
+ pathlib.Path(__file__).resolve().parents[2] / "serve" / "scripts" / "seed_fixtures" / "react_committee.json"
266
+ )
251
267
 
252
268
 
253
269
  def _notebook_trace(now: datetime.datetime) -> dict:
@@ -384,7 +400,157 @@ def _notebook_trace(now: datetime.datetime) -> dict:
384
400
  "adapter_type": "ChatAdapter",
385
401
  "request_preview": "Colorado 2026 U.S. House elections (Ballotpedia markdown)",
386
402
  "response_preview": f"{state}: {districts} districts",
387
- "project_tags": _SEED_TAGS + ["notebook-demo"],
403
+ "project_tags": _SEED_TAGS + [_GOLDEN_PATH, "notebook-demo"],
404
+ "total_tokens": total_tokens,
405
+ "span_count": len(spans),
406
+ "spans": spans,
407
+ }
408
+
409
+
410
+ def _notebook_trace_in_progress(now: datetime.datetime) -> dict:
411
+ """A SECOND DistrictCounter RLM trace, frozen mid-flight as an `in_progress`
412
+ live-notebook (CMP-634) so the review app demonstrates the live-streaming
413
+ autofill alongside the completed run from _notebook_trace.
414
+
415
+ Built from the same fixture but keeping only the first half of the REPL
416
+ turns, with no authoritative trajectory on the root and no answer fields —
417
+ so serve's ParseNotebook reconstructs a *partial* notebook from the Predict
418
+ turn children, and status "in_progress" (+ root span_type "rlm") renders it
419
+ as the "running" notebook with the live spinner (serve traces.templ
420
+ notebookPanel: Partial && Live). The trace + root span carry no end_time
421
+ (NULL via the batch path's nullableTime), the shape a real live trace has.
422
+
423
+ start_time is `now` so it reads as freshly in-flight. A background sweep in
424
+ serve settles stale in_progress traces after ~15m; review apps are seeded
425
+ fresh per PR, so this trace is comfortably inside that window when a reviewer
426
+ opens it — no sweep change needed. Posted via /api/v1/traces/batch, whose
427
+ InsertWithSpans preserves an explicit "in_progress" status (detectIncomplete
428
+ is a no-op unless status == "ok") and stores a null end_time.
429
+ """
430
+ fx = json.loads(_RLM_FIXTURE.read_text())
431
+ trajectory = fx["trajectory"]
432
+ turns = fx["turns"]
433
+ model = fx["model_name"]
434
+ trace_id = str(uuid.uuid4())
435
+ # Keep the first half of the turns; the run is "still going".
436
+ kept = max(1, len(trajectory) // 2)
437
+
438
+ def at(ns: int) -> str:
439
+ return (now + datetime.timedelta(microseconds=ns / 1000)).isoformat()
440
+
441
+ ctx = fx["inputs"].get("context")
442
+ context_str = ctx.get("_preview", "") if isinstance(ctx, dict) else (ctx or "")
443
+
444
+ root_id = str(uuid.uuid4())
445
+ # Root RLM span: in_progress, NO end_time, and NO outputs.trajectory — an
446
+ # authoritative trajectory would make serve build a *complete* notebook; its
447
+ # absence routes ParseNotebook to the partial (running) reconstruction.
448
+ spans: list[dict] = [
449
+ {
450
+ "span_id": root_id,
451
+ "trace_id": trace_id,
452
+ "parent_span_id": None,
453
+ "name": "RLM.forward",
454
+ "span_type": "rlm",
455
+ "start_time": at(0),
456
+ # The span status CHECK allows only unset/ok/error (in_progress is a
457
+ # trace-level status). A still-running root span is "unset" — the
458
+ # trace-level "in_progress" below is what drives the live notebook.
459
+ "status": "unset",
460
+ "module_type": "RLM",
461
+ "module_name": "RLM",
462
+ "signature_name": fx["signature_name"],
463
+ "program_signature_hash": fx["program_signature_hash"],
464
+ "signature_instructions": fx["instructions"],
465
+ "signature_input_fields": fx["input_fields"],
466
+ "signature_output_fields": fx["output_fields"],
467
+ "inputs": json.dumps({"context": context_str}),
468
+ },
469
+ ]
470
+
471
+ def adapter_span(parent: str, name: str, span_type: str, start: int, dur: int) -> dict:
472
+ return {
473
+ "span_id": str(uuid.uuid4()),
474
+ "trace_id": trace_id,
475
+ "parent_span_id": parent,
476
+ "name": name,
477
+ "span_type": span_type,
478
+ "start_time": at(start),
479
+ "end_time": at(start + dur),
480
+ "duration_ns": dur,
481
+ "status": "ok",
482
+ }
483
+
484
+ gap_ns = 50_000_000
485
+ cursor = 0
486
+ total_tokens = 0
487
+ for turn in range(kept):
488
+ step = trajectory[turn]
489
+ t = turns[turn]
490
+ p_start = cursor
491
+ p_end = p_start + t["predict_ns"]
492
+ predict_id = str(uuid.uuid4())
493
+ spans.append(
494
+ {
495
+ "span_id": predict_id,
496
+ "trace_id": trace_id,
497
+ "parent_span_id": root_id,
498
+ "name": "Predict.forward",
499
+ "span_type": "predict",
500
+ "module_type": "Predict",
501
+ "module_name": "Predict",
502
+ "start_time": at(p_start),
503
+ "end_time": at(p_end),
504
+ "duration_ns": t["predict_ns"],
505
+ "status": "ok",
506
+ "outputs": {"reasoning": step["reasoning"], "code": step["code"]},
507
+ }
508
+ )
509
+ spans.append(adapter_span(predict_id, "ChatAdapter.format", "adapter_format", p_start, t["format_ns"]))
510
+ lm_start = p_start + t["format_ns"]
511
+ lm_tokens = t["lm_tokens"]
512
+ completion = lm_tokens // 10
513
+ prompt = lm_tokens - completion
514
+ total_tokens += lm_tokens
515
+ spans.append(
516
+ {
517
+ "span_id": str(uuid.uuid4()),
518
+ "trace_id": trace_id,
519
+ "parent_span_id": predict_id,
520
+ "name": "LM.__call__",
521
+ "span_type": "lm_call",
522
+ "start_time": at(lm_start),
523
+ "end_time": at(lm_start + t["lm_ns"]),
524
+ "duration_ns": t["lm_ns"],
525
+ "status": "ok",
526
+ "model_name": model,
527
+ "model_provider": "bedrock",
528
+ "prompt_tokens": prompt,
529
+ "completion_tokens": completion,
530
+ "total_tokens": lm_tokens,
531
+ }
532
+ )
533
+ spans.append(
534
+ adapter_span(predict_id, "ChatAdapter.parse", "adapter_parse", p_end - t["parse_ns"], t["parse_ns"])
535
+ )
536
+ cursor = p_end + gap_ns
537
+
538
+ # No end_time / duration_ns at the trace level either — a live trace has not
539
+ # closed. status "in_progress" so serve renders the running notebook.
540
+ return {
541
+ "trace_id": trace_id,
542
+ "start_time": at(0),
543
+ "status": "in_progress",
544
+ "program_name": "RLM",
545
+ "program_version": "0.1.0",
546
+ "program_signature_hash": fx["program_signature_hash"],
547
+ "signature_name": fx["signature_name"],
548
+ "signature_input_fields": fx["input_fields"],
549
+ "signature_output_fields": fx["output_fields"],
550
+ "adapter_type": "ChatAdapter",
551
+ "request_preview": "Colorado 2026 U.S. House elections (Ballotpedia markdown)",
552
+ "response_preview": "",
553
+ "project_tags": _SEED_TAGS + [_GOLDEN_PATH, "notebook-demo", "in-progress"],
388
554
  "total_tokens": total_tokens,
389
555
  "span_count": len(spans),
390
556
  "spans": spans,
@@ -514,6 +680,85 @@ def _stamp_deployments(traces: list[dict], dep_for) -> None:
514
680
  trace["deployment_id"] = dep
515
681
 
516
682
 
683
+ # The review app's bootstrap admin (serve devEasySeedUser): admin@review.local.
684
+ # Its password is the review release's fixed default — release_serve_review.py's
685
+ # DEFAULT_SEED_ADMIN_PASSWORD, which the review flow never overrides — so the
686
+ # default below matches the deployed password with no workflow env needed.
687
+ # CMPND_SERVE_SEED_ADMIN_PASSWORD stays an override hook for the day that
688
+ # password is ever customized.
689
+ _ADMIN_EMAIL = "admin@review.local"
690
+ _ADMIN_PASSWORD_DEFAULT = "review-app-admin-password"
691
+ _CSRF_INPUT_RE = re.compile(r'name="csrf_token"\s+value="([^"]+)"')
692
+ _CSRF_META_RE = re.compile(r'name="csrf-token"\s+content="([^"]+)"')
693
+
694
+
695
+ def _display_program_hash(signed_hash: int) -> str:
696
+ """The 16-hex display form of a signed-int64 program_signature_hash — the
697
+ shape the /ui/programs/{hash}/pin route parses (ids.FormatProgramHashInt64
698
+ in serve: fmt.Sprintf("%016x", uint64(n)))."""
699
+ return format(signed_hash & 0xFFFFFFFFFFFFFFFF, "016x")
700
+
701
+
702
+ def _scrape_csrf(html: str) -> str | None:
703
+ """Pull a CSRF token from a serve HTML page — the hidden `csrf_token` form
704
+ input, or the `<meta name="csrf-token">` fallback (both carry the per-session
705
+ token)."""
706
+ m = _CSRF_INPUT_RE.search(html) or _CSRF_META_RE.search(html)
707
+ return m.group(1) if m else None
708
+
709
+
710
+ def _pin_committee_program(endpoint: str, display_hash: str) -> None:
711
+ """Pin the completed committee program for the demo admin (CMP-634).
712
+
713
+ Pins are per-user web state: `program_pins` FKs the ingested `programs` row
714
+ and serve's Pin() is reachable ONLY through the session+CSRF web route
715
+ POST /ui/programs/{hash}/pin — the API key (bearer, no cookies) can't hit it.
716
+ So we drive the browser flow with a cookie jar: log in as the bootstrap admin,
717
+ harvest a fresh CSRF token from the program page, then POST the pin. Wholly
718
+ non-gating — any failure logs and returns, like the rest of the seed.
719
+
720
+ Not exercisable against local dev (empty-password admin); it runs for real in
721
+ the review app, whose admin password is _ADMIN_PASSWORD_DEFAULT (the review
722
+ release's fixed default).
723
+ """
724
+ password = os.getenv("CMPND_SERVE_SEED_ADMIN_PASSWORD") or _ADMIN_PASSWORD_DEFAULT
725
+ jar = http.cookiejar.CookieJar()
726
+ opener = urllib.request.build_opener(urllib.request.HTTPCookieProcessor(jar))
727
+
728
+ def get(path: str) -> str:
729
+ req = urllib.request.Request(endpoint + path, method="GET")
730
+ with opener.open(req, timeout=30) as resp: # noqa: S310 - fixed endpoint from env
731
+ return resp.read().decode("utf-8", "replace")
732
+
733
+ def post(path: str, fields: dict[str, str]) -> int:
734
+ data = urllib.parse.urlencode(fields).encode("utf-8")
735
+ req = urllib.request.Request(
736
+ endpoint + path,
737
+ data=data,
738
+ method="POST",
739
+ headers={"Content-Type": "application/x-www-form-urlencoded"},
740
+ )
741
+ with opener.open(req, timeout=30) as resp: # noqa: S310 - fixed endpoint from env
742
+ return resp.status
743
+
744
+ # 1) GET the login page → session cookie + a CSRF token to submit with login.
745
+ login_csrf = _scrape_csrf(get("/auth/login"))
746
+ if not login_csrf:
747
+ log.warning("pin: no CSRF token on /auth/login; skipping pin")
748
+ return
749
+ # 2) Log in. serve rotates the CSRF token on login, so the token above is
750
+ # single-use for this POST only.
751
+ post("/auth/login", {"email": _ADMIN_EMAIL, "password": password, "csrf_token": login_csrf})
752
+ # 3) Harvest a FRESH CSRF token from the (now authenticated) program page.
753
+ pin_csrf = _scrape_csrf(get(f"/ui/programs/{display_hash}"))
754
+ if not pin_csrf:
755
+ log.warning("pin: no CSRF token on program page (login may have failed); skipping pin")
756
+ return
757
+ # 4) POST the pin.
758
+ status = post(f"/ui/programs/{display_hash}/pin", {"csrf_token": pin_csrf})
759
+ log.info("pinned committee program %s (status %s)", display_hash, status)
760
+
761
+
517
762
  def main() -> int:
518
763
  endpoint = (os.getenv("CMPND_E2E_ENDPOINT") or "").rstrip("/")
519
764
  api_key = os.getenv("CMPND_E2E_API_KEY") or ""
@@ -579,6 +824,20 @@ def main() -> int:
579
824
  n_spans += 1
580
825
  log.info("seeded %d traces, %d span-batches, %d stat rows -> %s", n_traces, n_spans, n_stats, new_op)
581
826
 
827
+ # 2b) Pin the completed committee program for the demo admin (CMP-634) so
828
+ # the sidebar shows a pinned program. Runs now that the trace ingest has
829
+ # upserted the programs row the pin FKs. Self-contained try/except so the
830
+ # login+web-route flow (the only path to a per-user pin) can never break
831
+ # the seed.
832
+ try:
833
+ committee_hash = opt["payload"].get("program_signature_hash")
834
+ if committee_hash is not None:
835
+ _pin_committee_program(endpoint, _display_program_hash(int(committee_hash)))
836
+ else:
837
+ log.warning("pin: optimization payload has no program_signature_hash; skipping pin")
838
+ except Exception as exc: # noqa: BLE001 - non-gating, like the rest of the seed
839
+ log.warning("pin committee program failed (non-gating): %s", exc)
840
+
582
841
  # 3) Plant the interrupted twin of the completed run — a rich `running`
583
842
  # row frozen right after iteration 2 (real iterations/candidates/scores
584
843
  # + its own linked traces), sharing the completed run's `log_dir` so the
@@ -651,6 +910,14 @@ def main() -> int:
651
910
  code, _ = _post(endpoint, "/api/v1/traces/batch", notebook_batch, api_key)
652
911
  log.info("seeded RLM notebook trace (status %s)", code)
653
912
 
913
+ # 4a) Plant a SECOND DistrictCounter RLM trace, frozen mid-flight as an
914
+ # in_progress live notebook (CMP-634) so the demo shows the live-streaming
915
+ # autofill next to the completed run above. No end_time, partial trajectory,
916
+ # status in_progress → serve renders the "running" notebook. Left unstamped
917
+ # (a live run predates any deployment link).
918
+ code, _ = _post(endpoint, "/api/v1/traces/batch", [_notebook_trace_in_progress(now)], api_key)
919
+ log.info("seeded in-progress RLM notebook trace (status %s)", code)
920
+
654
921
  # 4b) Plant the two CMP-187 tool-span demo traces so every review app
655
922
  # shows RLM tool spans (SourceCategorizer, with a vision sub-LLM nested
656
923
  # under Tool.inspect_page) and the function-tool vs module-tool
@@ -683,6 +950,22 @@ def main() -> int:
683
950
  )
684
951
  else:
685
952
  log.warning("skipping ToolKindDemo seed: fixture missing at %s", _TOOL_KIND_DEMO_FIXTURE)
953
+ # A dspy.ReAct(ExtractCommittee) trace (CMP-634): reasons, calls a
954
+ # `lookup_disclaimer` tool to pull the "Paid for by" line, then extracts
955
+ # the committee — the tool-span ReAct exemplar the golden-path demo lacked.
956
+ if _REACT_COMMITTEE_FIXTURE.exists():
957
+ tool_demos.append(
958
+ _replay_flat_fixture(
959
+ _REACT_COMMITTEE_FIXTURE,
960
+ now,
961
+ "ReAct",
962
+ _SEED_TAGS + [_GOLDEN_PATH, "react-demo"],
963
+ "committee fundraising email (ReAct + lookup_disclaimer tool)",
964
+ "NATIONAL REPUBLICAN SENATORIAL COMMITTEE",
965
+ )
966
+ )
967
+ else:
968
+ log.warning("skipping ReAct committee seed: fixture missing at %s", _REACT_COMMITTEE_FIXTURE)
686
969
  if tool_demos:
687
970
  _stamp_deployments(tool_demos, dep_for)
688
971
  code, _ = _post(endpoint, "/api/v1/traces/batch", tool_demos, api_key)
@@ -47,10 +47,22 @@ _DATA_DIR = pathlib.Path(__file__).resolve().parent / "data"
47
47
  # Bedrock's OpenAI-compatible surface, where they are `404 model_not_found` (they live on
48
48
  # the native Converse surface litellm translates to), so a run against that plane, or
49
49
  # against any other provider, has to be able to say so without editing this file.
50
- STUDENT_1B = os.getenv("PLATFORM_STRESS_MODEL") or "bedrock/us.amazon.nova-micro-v1:0" # tiny/cheap; no lift assertion
51
- STUDENT_3B = os.getenv("PLATFORM_STUDENT_MODEL") or "bedrock/google.gemma-3-4b-it" # weak-with-headroom; gepa student
50
+ #
51
+ # The DEPLOYED_ variables, because that is what every model here is for: this module's
52
+ # programs are built to be deployed and executed by a data plane, so the plane's provider
53
+ # constraint is theirs. The un-prefixed `PLATFORM_*_MODEL` names configure LMs that a test
54
+ # process calls directly, and reading those here is what made a wasm-plane run also rewire
55
+ # tests that never touch a plane. `_provider` (which this module cannot import — it
56
+ # predates the stress suite's resolver and is shared with a lane that has no wasm plane)
57
+ # carries the full note.
58
+ STUDENT_1B = (
59
+ os.getenv("PLATFORM_DEPLOYED_STRESS_MODEL") or "bedrock/us.amazon.nova-micro-v1:0"
60
+ ) # tiny/cheap; no lift assertion
61
+ STUDENT_3B = (
62
+ os.getenv("PLATFORM_DEPLOYED_STUDENT_MODEL") or "bedrock/google.gemma-3-4b-it"
63
+ ) # weak-with-headroom; gepa student
52
64
  REFLECTION_MODEL = (
53
- os.getenv("PLATFORM_REFLECTION_MODEL") or "bedrock/us.meta.llama3-3-70b-instruct-v1:0"
65
+ os.getenv("PLATFORM_DEPLOYED_REFLECTION_MODEL") or "bedrock/us.meta.llama3-3-70b-instruct-v1:0"
54
66
  ) # gepa reflection (resolved server-side)
55
67
 
56
68
 
@@ -89,8 +101,11 @@ def lm_api_base() -> str | None:
89
101
  ``lm_api_host``, which seeds the org's provider key, which is what the plane later
90
102
  resolves into ``inference.student.api_base``. Without it a run there fails on the
91
103
  broker's own "no api_base" refusal.
104
+
105
+ ``PLATFORM_DEPLOYED_LM_API_BASE``, matching the model ids above: this base is declared
106
+ on a program the PLANE runs, so it must not also redirect a test's own litellm calls.
92
107
  """
93
- return os.getenv("PLATFORM_LM_API_BASE") or None
108
+ return os.getenv("PLATFORM_DEPLOYED_LM_API_BASE") or None
94
109
 
95
110
 
96
111
  def _lm_kwargs() -> dict[str, Any]: