cmpnd 0.7.1__tar.gz → 0.7.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {cmpnd-0.7.1 → cmpnd-0.7.2}/PKG-INFO +2 -1
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/_gepa_patch.py +9 -2
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/optimizers/gepa.py +6 -2
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/optimizers/gepa_callback.py +1 -1
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/optimizers/tracker.py +11 -4
- {cmpnd-0.7.1 → cmpnd-0.7.2}/pyproject.toml +38 -1
- {cmpnd-0.7.1 → cmpnd-0.7.2}/scripts/seed_review_run.py +290 -7
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/committee_task.py +19 -4
- cmpnd-0.7.2/tests/stress/_provider.py +276 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/stress/test_deploy_programs.py +38 -8
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/stress/test_optimize_loop.py +8 -1
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/stress/test_optimize_loop_wasm.py +71 -24
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/stress/test_optimize_stress.py +6 -1
- cmpnd-0.7.2/tests/stress/test_provider_roles.py +221 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/stress/test_real_provider_twins.py +19 -2
- cmpnd-0.7.2/tests/stress/test_rlm_wasm.py +411 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_clustering/conftest.py +2 -9
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_clustering/test_edge_cases.py +2 -7
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_clustering/test_hash_consistency.py +4 -11
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_clustering/test_type_conversion.py +3 -8
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_deploy.py +50 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_optimize_terminal_failure.py +78 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_seed_review_run.py +10 -4
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_tracked_gepa.py +5 -1
- {cmpnd-0.7.1 → cmpnd-0.7.2}/uv.lock +58 -94
- cmpnd-0.7.1/tests/stress/_provider.py +0 -91
- {cmpnd-0.7.1 → cmpnd-0.7.2}/.gitignore +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/CLAUDE.md +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/README.md +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/__init__.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/_program_patch.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/_rlm_patch.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/callback.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/cli/__init__.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/cli/api_client.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/cli/commands/__init__.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/cli/commands/admin.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/cli/commands/auth.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/cli/commands/datasets.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/cli/commands/deployments.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/cli/commands/evals.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/cli/commands/login.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/cli/commands/optimizations.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/cli/commands/org.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/cli/commands/programs.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/cli/commands/top.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/cli/commands/traces.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/cli/credentials.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/cli/shell.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/cli/trace_render.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/cli/verbs.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/configuration.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/context.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/dataset_sync.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/datasets.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/decorators.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/deployment.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/eval_handler.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/execution.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/exporter.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/exporter_http.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/helpers.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/identity.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/ir/__init__.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/ir/decode.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/ir/encode.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/ir/hash.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/ir/program.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/ir/reflection.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/ir/types.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/ir/xxh64.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/models.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/optimization.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/optimizers/__init__.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/optimizers/resume.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/cmpnd/packaging.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/docs/2026-04-02-code-review.md +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/docs/plans/2026-04-05-train-val-dataset-capture-design.md +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/docs/plans/2026-04-05-train-val-dataset-capture.md +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/docs/sdk-instrumentation.md +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/examples/local_ollama.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/scripts/capture_review_run.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/scripts/fixtures/review_run.json.gz +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/scripts/gepa_resume_bug.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/stubs/dspy/__init__.pyi +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/stubs/dspy/primitives/__init__.pyi +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/stubs/dspy/primitives/prediction.pyi +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/stubs/dspy/utils/__init__.pyi +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/stubs/dspy/utils/callback.pyi +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/__init__.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/_exporter_helpers.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/_identity_helpers.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/__init__.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/conftest.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/data/committee_train.ndjson.gz +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/data/committee_val.ndjson.gz +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/fake_lm.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/helpers.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/test_capture_prompts.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/test_composite_module.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/test_dataset_auto_creation.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/test_deploy_execute.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/test_deploy_programs.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/test_edge_cases.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/test_example_hash.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/test_gepa_optimization.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/test_lm_accuracy.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/test_module_tracing.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/test_optimization_lineage.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/test_optimization_metric_identity.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/test_optimize_failures.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/test_optimize_loop.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/test_optimize_review_committee.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/test_optimize_server_owned_completion.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/test_parallel_export.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/test_project_attribution.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/test_react_tool_spans.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/test_save_load.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/test_trace_structure.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/e2e/test_user_attribution.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/identity_vectors/01_predict_qa.json +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/identity_vectors/02_predict_renamed_field.json +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/identity_vectors/03_chain_of_thought_qa.json +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/identity_vectors/04_predict_richer_types.json +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/identity_vectors/05_composite_two_predicts.json +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/identity_vectors/06_multichaincomparison.json +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/identity_vectors/07_program_of_thought.json +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/identity_vectors/08_opaque_module.json +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/identity_vectors/09_predict_pydantic_field.json +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/identity_vectors/10_react.json +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/identity_vectors/11_refine.json +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/identity_vectors/12_retrieve.json +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/identity_vectors/13_knn.json +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/identity_vectors/14_parallel.json +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/identity_vectors/README.md +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/integration/__init__.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/integration/conftest.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/integration/test_cli_deploy_smoketest.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/integration/test_logs_renders_traces.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/integration/test_optimizing_progress.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/integration/test_unhappy_path.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/parity/__init__.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/parity/conftest.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/parity/seed_parity.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/parity/test_auth.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/parity/test_datasets.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/parity/test_deployments.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/parity/test_evals.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/parity/test_health.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/parity/test_modules.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/parity/test_optimizations.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/parity/test_programs.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/parity/test_signatures.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/parity/test_trace_streaming.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/parity/test_traces.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/stress/__init__.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/stress/_reliability.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/stress/mint_bedrock_token.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/stress/test_multi_client_stress.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/stress/test_optimize_client_committee.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/stress/test_reliability.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/stress/test_reliability_classify.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_callback.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_cli/__init__.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_cli/conftest.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_cli/test_admin.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_cli/test_api_client.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_cli/test_auth.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_cli/test_credentials.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_cli/test_datasets.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_cli/test_deployments.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_cli/test_evals.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_cli/test_login.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_cli/test_main.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_cli/test_optimizations.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_cli/test_org.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_cli/test_parity.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_cli/test_programs.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_cli/test_shell.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_cli/test_top.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_cli/test_traces.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_clustering/__init__.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_cmp551_grouping_identity.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_configuration.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_context.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_context_propagation.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_dataset_polling.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_dataset_sync.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_datasets.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_decorators.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_eval.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_eval_start.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_exception_paths.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_execute.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_exporter.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_exporter_real.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_fake_lm.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_gepa_callback.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_gepa_integration.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_gepa_resume.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_identity.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_import_weight.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_integration.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_ir_encoder_properties.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_ir_golden_vectors.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_ir_roundtrip.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_models.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_optimization_progress.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_optimization_start.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_optimize.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_optimize_lift.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_packaging.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_packaging_callable_source.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_packaging_program_source.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_program_lineage_persist.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_project_payload.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_reflector_properties.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_reflector_smoke.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_repro_reactv2_bugs.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_rlm_patch.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_schemas.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_stats_export.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_trace_render.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_tracker_enrich.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_truncation.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/tests/test_xxh64.py +0 -0
- {cmpnd-0.7.1 → cmpnd-0.7.2}/todo.md +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: cmpnd
|
|
3
|
-
Version: 0.7.
|
|
3
|
+
Version: 0.7.2
|
|
4
4
|
Summary: DSPy observability and deployment SDK for cmpnd
|
|
5
5
|
Project-URL: Homepage, https://cmpnd.ai
|
|
6
6
|
Author-email: cmpnd <hello@cmpnd.ai>
|
|
@@ -27,6 +27,7 @@ Requires-Dist: basedpyright>=1.39; extra == 'dev'
|
|
|
27
27
|
Requires-Dist: boto3>=1.34; extra == 'dev'
|
|
28
28
|
Requires-Dist: hypothesis>=6.100; extra == 'dev'
|
|
29
29
|
Requires-Dist: jsonschema>=4.0; extra == 'dev'
|
|
30
|
+
Requires-Dist: numpy>=1.26; extra == 'dev'
|
|
30
31
|
Requires-Dist: pytest-asyncio>=0.23; extra == 'dev'
|
|
31
32
|
Requires-Dist: pytest-rerunfailures>=14.0; extra == 'dev'
|
|
32
33
|
Requires-Dist: pytest-xdist>=3.0; extra == 'dev'
|
|
@@ -64,7 +64,7 @@ def patch_gepa() -> None:
|
|
|
64
64
|
# This ensures each optimization gets its own run_id, even when
|
|
65
65
|
# the same GEPA instance is reused (e.g., optimizing an already-optimized program).
|
|
66
66
|
trk = tracker.OptimizationTracker(
|
|
67
|
-
optimizer_type="
|
|
67
|
+
optimizer_type="gepa",
|
|
68
68
|
program_name="",
|
|
69
69
|
)
|
|
70
70
|
callback = gepa_callback.CmpndGEPACallback(trk, defer_finalize=True)
|
|
@@ -157,7 +157,14 @@ def patch_gepa() -> None:
|
|
|
157
157
|
|
|
158
158
|
return result
|
|
159
159
|
|
|
160
|
-
|
|
160
|
+
# `BaseException`, because the invariant does not get to have an exception: a
|
|
161
|
+
# run that ends by ANY route must transition the row start() sent, or the
|
|
162
|
+
# program page shows a perpetual in-progress optimization. A backend that
|
|
163
|
+
# rejected a model request raises a failure that deliberately does NOT derive
|
|
164
|
+
# from `Exception` — that is the only way past gepa's and dspy's own blanket
|
|
165
|
+
# `except Exception` clauses — so narrowing this one reopens exactly the bug
|
|
166
|
+
# this block exists for. Re-raised either way, so widening swallows nothing.
|
|
167
|
+
except BaseException as e:
|
|
161
168
|
if callback is not None: # pyright: ignore[reportUnnecessaryComparison]
|
|
162
169
|
# Flush a terminal 'failed', not just in-memory state — else the
|
|
163
170
|
# running row start() sent lingers forever.
|
|
@@ -152,7 +152,7 @@ class TrackedGEPA:
|
|
|
152
152
|
|
|
153
153
|
# Initialize optimization tracking
|
|
154
154
|
self._tracker = tracker.OptimizationTracker(
|
|
155
|
-
optimizer_type="
|
|
155
|
+
optimizer_type="gepa",
|
|
156
156
|
program_name=student.__class__.__name__,
|
|
157
157
|
config=config,
|
|
158
158
|
model_name=model_name,
|
|
@@ -218,7 +218,11 @@ class TrackedGEPA:
|
|
|
218
218
|
|
|
219
219
|
return result
|
|
220
220
|
|
|
221
|
-
|
|
221
|
+
# `BaseException` for the reason `_gepa_patch._patched_compile` documents: the
|
|
222
|
+
# row must transition however the run ends, and a rejected model request
|
|
223
|
+
# reaches here as something that is not an `Exception`, by design. Re-raised
|
|
224
|
+
# either way.
|
|
225
|
+
except BaseException as e:
|
|
222
226
|
if self._tracker:
|
|
223
227
|
# Flush a terminal 'failed', not just in-memory state — else the
|
|
224
228
|
# running row start() sent lingers forever.
|
|
@@ -5,7 +5,7 @@ monkey-patching or wrapper classes. Replaces the monkey-patched
|
|
|
5
5
|
InstrumentedDspyAdapter approach used by TrackedGEPA.
|
|
6
6
|
|
|
7
7
|
Usage:
|
|
8
|
-
tracker = OptimizationTracker(optimizer_type="
|
|
8
|
+
tracker = OptimizationTracker(optimizer_type="gepa", ...)
|
|
9
9
|
callback = CmpndGEPACallback(tracker)
|
|
10
10
|
gepa = dspy.GEPA(gepa_kwargs={"callbacks": [callback]})
|
|
11
11
|
"""
|
|
@@ -199,7 +199,7 @@ class OptimizationTracker:
|
|
|
199
199
|
|
|
200
200
|
Usage:
|
|
201
201
|
tracker = OptimizationTracker(
|
|
202
|
-
optimizer_type="
|
|
202
|
+
optimizer_type="gepa",
|
|
203
203
|
program_name="MyProgram",
|
|
204
204
|
config={"auto": "light", "seed": 42}
|
|
205
205
|
)
|
|
@@ -634,14 +634,21 @@ class OptimizationTracker:
|
|
|
634
634
|
except Exception as e:
|
|
635
635
|
logger.warning(f"Failed to capture compiled program state: {e}")
|
|
636
636
|
|
|
637
|
-
def set_error(self, exception:
|
|
638
|
-
"""Mark the optimization run as failed (in-memory only).
|
|
637
|
+
def set_error(self, exception: BaseException) -> None:
|
|
638
|
+
"""Mark the optimization run as failed (in-memory only).
|
|
639
|
+
|
|
640
|
+
``BaseException`` and not ``Exception``: a backend that rejected a model request
|
|
641
|
+
reports it as a failure that deliberately does not derive from ``Exception``,
|
|
642
|
+
since that is the only thing gepa's and dspy's blanket ``except Exception``
|
|
643
|
+
clauses do not swallow. A run that ended is a run that has to be recorded as
|
|
644
|
+
ended, whichever base class carried the reason.
|
|
645
|
+
"""
|
|
639
646
|
self.status = "failed"
|
|
640
647
|
self.error_message = str(exception)
|
|
641
648
|
self.end_time = datetime.now(timezone.utc)
|
|
642
649
|
logger.error(f"Optimization failed: {exception}")
|
|
643
650
|
|
|
644
|
-
def finalize_error(self, exception:
|
|
651
|
+
def finalize_error(self, exception: BaseException) -> None:
|
|
645
652
|
"""Mark the run failed AND flush the terminal 'failed' to the backend.
|
|
646
653
|
|
|
647
654
|
The success path is finalize(); on an optimizer exception the run must
|
|
@@ -5,7 +5,7 @@ name = "cmpnd"
|
|
|
5
5
|
# it's new) — no git tags. Bump with `uv version --bump {minor,patch}` or by
|
|
6
6
|
# hand. Runtime `cmpnd.__version__` reads installed metadata, so it tracks
|
|
7
7
|
# this automatically. See docs/plans/2026-07-02-sdk-pypi-tag-free-publish.md.
|
|
8
|
-
version = "0.7.
|
|
8
|
+
version = "0.7.2"
|
|
9
9
|
description = "DSPy observability and deployment SDK for cmpnd"
|
|
10
10
|
readme = "README.md"
|
|
11
11
|
requires-python = ">=3.10"
|
|
@@ -58,6 +58,10 @@ dev = [
|
|
|
58
58
|
# litellm's bedrock/ provider imports boto3 in-process, so the client-side
|
|
59
59
|
# Bedrock stress tests (committee twins) need it in the test env.
|
|
60
60
|
"boto3>=1.34",
|
|
61
|
+
# test_reflector_smoke.py builds numpy fixtures. dspy <=3.2.x required numpy
|
|
62
|
+
# unconditionally so it rode in transitively; dspy 3.3.0 moved numpy behind a
|
|
63
|
+
# `numpy` extra, so declare it directly as the test dependency it always was.
|
|
64
|
+
"numpy>=1.26",
|
|
61
65
|
]
|
|
62
66
|
# Deps for `deploy/deploy/stub/`, the local FastAPI server that emulates
|
|
63
67
|
# the AWS deploy stack. The stub itself ships in the cmpnd-deploy
|
|
@@ -92,6 +96,39 @@ build-backend = "hatchling.build"
|
|
|
92
96
|
[tool.hatch.build.targets.wheel]
|
|
93
97
|
packages = ["cmpnd"]
|
|
94
98
|
|
|
99
|
+
[tool.pytest.ini_options]
|
|
100
|
+
# Deliberately minimal: `markers` and `--strict-markers`, nothing else. Setting
|
|
101
|
+
# `testpaths` or a broader `addopts` here would change how every existing
|
|
102
|
+
# invocation resolves (CI lanes, `uv run pytest tests/ -k "not e2e"`, a bare
|
|
103
|
+
# `pytest` from a subdirectory), and this file's job is only to make the marker
|
|
104
|
+
# vocabulary explicit.
|
|
105
|
+
#
|
|
106
|
+
# --strict-markers turns a typo'd marker into an error instead of a silent
|
|
107
|
+
# no-op. That matters most for `forall_workers`: it SELECTS the merge-queue's
|
|
108
|
+
# wasm lane, so a misspelling there would quietly select nothing and report a
|
|
109
|
+
# green lane that ran zero tests. `asyncio` and `flaky` are registered by
|
|
110
|
+
# pytest-asyncio / pytest-rerunfailures, so they are not listed here.
|
|
111
|
+
addopts = "--strict-markers"
|
|
112
|
+
markers = [
|
|
113
|
+
# Tests that must hold for EVERY worker backend — i.e. that provision,
|
|
114
|
+
# execute or optimize through a data plane, and so can distinguish the
|
|
115
|
+
# Lambda plane from the wasm plane from the on-prem worker. This is the
|
|
116
|
+
# selector for the per-plane stress lanes: the set worth running against a
|
|
117
|
+
# second worker is exactly the set claimed to hold for all of them.
|
|
118
|
+
#
|
|
119
|
+
# A stress test that deploys or executes gets this marker. Its absence on
|
|
120
|
+
# such a test is a bug; its PRESENCE on a test no worker can affect (one
|
|
121
|
+
# that only exercises client-side tracing, or serve's own health) is worse,
|
|
122
|
+
# because it costs a second real-provider run to prove nothing.
|
|
123
|
+
# See docs/plans/2026-07-29-wasm-stress-lane.md.
|
|
124
|
+
"forall_workers: must hold for every worker backend (reaches a data plane)",
|
|
125
|
+
# Pre-existing vocabulary, registered here because --strict-markers now
|
|
126
|
+
# requires it — not introduced by this block.
|
|
127
|
+
"parity: CLI dual-front-end parity (argparse vs. interactive shell)",
|
|
128
|
+
"e2e: needs a live platform; excluded from the fast lane via -k 'not e2e'",
|
|
129
|
+
"integration: crosses a component boundary (SDK ↔ serve ↔ worker)",
|
|
130
|
+
]
|
|
131
|
+
|
|
95
132
|
[tool.ruff]
|
|
96
133
|
line-length = 120
|
|
97
134
|
target-version = "py310"
|
|
@@ -33,12 +33,14 @@ from __future__ import annotations
|
|
|
33
33
|
|
|
34
34
|
import datetime
|
|
35
35
|
import gzip
|
|
36
|
+
import http.cookiejar
|
|
36
37
|
import json
|
|
37
38
|
import logging
|
|
38
39
|
import os
|
|
39
40
|
import pathlib
|
|
40
41
|
import re
|
|
41
42
|
import urllib.error
|
|
43
|
+
import urllib.parse
|
|
42
44
|
import urllib.request
|
|
43
45
|
import uuid
|
|
44
46
|
|
|
@@ -46,13 +48,19 @@ logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(mess
|
|
|
46
48
|
log = logging.getLogger("seed_review_run")
|
|
47
49
|
|
|
48
50
|
_FIXTURE = pathlib.Path(__file__).resolve().parent / "fixtures" / "review_run.json.gz"
|
|
49
|
-
# Origin tag so seeded data is identifiable as such (never
|
|
50
|
-
# customer runs). Overrides whatever the captured fixture
|
|
51
|
-
# run also carries a `lifecycle-*` role tag (CMP-414) so its
|
|
52
|
-
# lifecycle demo is explicit — and so the failed run reads as
|
|
53
|
-
# an incident.
|
|
51
|
+
# Origin tag on ALL seeded rows so seeded data is identifiable as such (never
|
|
52
|
+
# confused with real customer runs). Overrides whatever the captured fixture
|
|
53
|
+
# carried. Each seeded run also carries a `lifecycle-*` role tag (CMP-414) so its
|
|
54
|
+
# place in the lifecycle demo is explicit — and so the failed run reads as
|
|
55
|
+
# intentional, not an incident.
|
|
54
56
|
_SEED_TAGS = ["seed-data"]
|
|
55
|
-
|
|
57
|
+
# `golden-path` marks ONLY the data that IS a DSPy-Meetup golden-path demo beat
|
|
58
|
+
# (CMP-634) — the completed committee run + its Predict traces, the ReAct
|
|
59
|
+
# committee trace, and the RLM notebook (completed + in-progress). It does NOT
|
|
60
|
+
# go on the lifecycle running/failed twins or the CMP-187 tool-demo traces, so
|
|
61
|
+
# `?tags=golden-path` surfaces exactly the demo tenant, nothing incidental.
|
|
62
|
+
_GOLDEN_PATH = "golden-path"
|
|
63
|
+
_TAGS_COMPLETE = _SEED_TAGS + [_GOLDEN_PATH, "lifecycle-complete"]
|
|
56
64
|
_TAGS_RUNNING = _SEED_TAGS + ["lifecycle-running"]
|
|
57
65
|
_TAGS_FAILED = _SEED_TAGS + ["lifecycle-failed"]
|
|
58
66
|
_UUID_RE = re.compile(r"[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}")
|
|
@@ -248,6 +256,14 @@ _SOURCE_CATEGORIZER_FIXTURE = (
|
|
|
248
256
|
_TOOL_KIND_DEMO_FIXTURE = (
|
|
249
257
|
pathlib.Path(__file__).resolve().parents[2] / "serve" / "scripts" / "seed_fixtures" / "tool_kind_demo.json"
|
|
250
258
|
)
|
|
259
|
+
# A dspy.ReAct(ExtractCommittee) run (email_body->committee) that reasons, calls
|
|
260
|
+
# a `lookup_disclaimer` tool to pull the "Paid for by" line, then extracts the
|
|
261
|
+
# committee (CMP-634 golden-path). Flat-replay format like the two fixtures
|
|
262
|
+
# above; program_name "ReAct" so the sidebar/Type badge reads ReAct. Loaded
|
|
263
|
+
# lazily so a missing file degrades to skipping that seed.
|
|
264
|
+
_REACT_COMMITTEE_FIXTURE = (
|
|
265
|
+
pathlib.Path(__file__).resolve().parents[2] / "serve" / "scripts" / "seed_fixtures" / "react_committee.json"
|
|
266
|
+
)
|
|
251
267
|
|
|
252
268
|
|
|
253
269
|
def _notebook_trace(now: datetime.datetime) -> dict:
|
|
@@ -384,7 +400,157 @@ def _notebook_trace(now: datetime.datetime) -> dict:
|
|
|
384
400
|
"adapter_type": "ChatAdapter",
|
|
385
401
|
"request_preview": "Colorado 2026 U.S. House elections (Ballotpedia markdown)",
|
|
386
402
|
"response_preview": f"{state}: {districts} districts",
|
|
387
|
-
"project_tags": _SEED_TAGS + ["notebook-demo"],
|
|
403
|
+
"project_tags": _SEED_TAGS + [_GOLDEN_PATH, "notebook-demo"],
|
|
404
|
+
"total_tokens": total_tokens,
|
|
405
|
+
"span_count": len(spans),
|
|
406
|
+
"spans": spans,
|
|
407
|
+
}
|
|
408
|
+
|
|
409
|
+
|
|
410
|
+
def _notebook_trace_in_progress(now: datetime.datetime) -> dict:
|
|
411
|
+
"""A SECOND DistrictCounter RLM trace, frozen mid-flight as an `in_progress`
|
|
412
|
+
live-notebook (CMP-634) so the review app demonstrates the live-streaming
|
|
413
|
+
autofill alongside the completed run from _notebook_trace.
|
|
414
|
+
|
|
415
|
+
Built from the same fixture but keeping only the first half of the REPL
|
|
416
|
+
turns, with no authoritative trajectory on the root and no answer fields —
|
|
417
|
+
so serve's ParseNotebook reconstructs a *partial* notebook from the Predict
|
|
418
|
+
turn children, and status "in_progress" (+ root span_type "rlm") renders it
|
|
419
|
+
as the "running" notebook with the live spinner (serve traces.templ
|
|
420
|
+
notebookPanel: Partial && Live). The trace + root span carry no end_time
|
|
421
|
+
(NULL via the batch path's nullableTime), the shape a real live trace has.
|
|
422
|
+
|
|
423
|
+
start_time is `now` so it reads as freshly in-flight. A background sweep in
|
|
424
|
+
serve settles stale in_progress traces after ~15m; review apps are seeded
|
|
425
|
+
fresh per PR, so this trace is comfortably inside that window when a reviewer
|
|
426
|
+
opens it — no sweep change needed. Posted via /api/v1/traces/batch, whose
|
|
427
|
+
InsertWithSpans preserves an explicit "in_progress" status (detectIncomplete
|
|
428
|
+
is a no-op unless status == "ok") and stores a null end_time.
|
|
429
|
+
"""
|
|
430
|
+
fx = json.loads(_RLM_FIXTURE.read_text())
|
|
431
|
+
trajectory = fx["trajectory"]
|
|
432
|
+
turns = fx["turns"]
|
|
433
|
+
model = fx["model_name"]
|
|
434
|
+
trace_id = str(uuid.uuid4())
|
|
435
|
+
# Keep the first half of the turns; the run is "still going".
|
|
436
|
+
kept = max(1, len(trajectory) // 2)
|
|
437
|
+
|
|
438
|
+
def at(ns: int) -> str:
|
|
439
|
+
return (now + datetime.timedelta(microseconds=ns / 1000)).isoformat()
|
|
440
|
+
|
|
441
|
+
ctx = fx["inputs"].get("context")
|
|
442
|
+
context_str = ctx.get("_preview", "") if isinstance(ctx, dict) else (ctx or "")
|
|
443
|
+
|
|
444
|
+
root_id = str(uuid.uuid4())
|
|
445
|
+
# Root RLM span: in_progress, NO end_time, and NO outputs.trajectory — an
|
|
446
|
+
# authoritative trajectory would make serve build a *complete* notebook; its
|
|
447
|
+
# absence routes ParseNotebook to the partial (running) reconstruction.
|
|
448
|
+
spans: list[dict] = [
|
|
449
|
+
{
|
|
450
|
+
"span_id": root_id,
|
|
451
|
+
"trace_id": trace_id,
|
|
452
|
+
"parent_span_id": None,
|
|
453
|
+
"name": "RLM.forward",
|
|
454
|
+
"span_type": "rlm",
|
|
455
|
+
"start_time": at(0),
|
|
456
|
+
# The span status CHECK allows only unset/ok/error (in_progress is a
|
|
457
|
+
# trace-level status). A still-running root span is "unset" — the
|
|
458
|
+
# trace-level "in_progress" below is what drives the live notebook.
|
|
459
|
+
"status": "unset",
|
|
460
|
+
"module_type": "RLM",
|
|
461
|
+
"module_name": "RLM",
|
|
462
|
+
"signature_name": fx["signature_name"],
|
|
463
|
+
"program_signature_hash": fx["program_signature_hash"],
|
|
464
|
+
"signature_instructions": fx["instructions"],
|
|
465
|
+
"signature_input_fields": fx["input_fields"],
|
|
466
|
+
"signature_output_fields": fx["output_fields"],
|
|
467
|
+
"inputs": json.dumps({"context": context_str}),
|
|
468
|
+
},
|
|
469
|
+
]
|
|
470
|
+
|
|
471
|
+
def adapter_span(parent: str, name: str, span_type: str, start: int, dur: int) -> dict:
|
|
472
|
+
return {
|
|
473
|
+
"span_id": str(uuid.uuid4()),
|
|
474
|
+
"trace_id": trace_id,
|
|
475
|
+
"parent_span_id": parent,
|
|
476
|
+
"name": name,
|
|
477
|
+
"span_type": span_type,
|
|
478
|
+
"start_time": at(start),
|
|
479
|
+
"end_time": at(start + dur),
|
|
480
|
+
"duration_ns": dur,
|
|
481
|
+
"status": "ok",
|
|
482
|
+
}
|
|
483
|
+
|
|
484
|
+
gap_ns = 50_000_000
|
|
485
|
+
cursor = 0
|
|
486
|
+
total_tokens = 0
|
|
487
|
+
for turn in range(kept):
|
|
488
|
+
step = trajectory[turn]
|
|
489
|
+
t = turns[turn]
|
|
490
|
+
p_start = cursor
|
|
491
|
+
p_end = p_start + t["predict_ns"]
|
|
492
|
+
predict_id = str(uuid.uuid4())
|
|
493
|
+
spans.append(
|
|
494
|
+
{
|
|
495
|
+
"span_id": predict_id,
|
|
496
|
+
"trace_id": trace_id,
|
|
497
|
+
"parent_span_id": root_id,
|
|
498
|
+
"name": "Predict.forward",
|
|
499
|
+
"span_type": "predict",
|
|
500
|
+
"module_type": "Predict",
|
|
501
|
+
"module_name": "Predict",
|
|
502
|
+
"start_time": at(p_start),
|
|
503
|
+
"end_time": at(p_end),
|
|
504
|
+
"duration_ns": t["predict_ns"],
|
|
505
|
+
"status": "ok",
|
|
506
|
+
"outputs": {"reasoning": step["reasoning"], "code": step["code"]},
|
|
507
|
+
}
|
|
508
|
+
)
|
|
509
|
+
spans.append(adapter_span(predict_id, "ChatAdapter.format", "adapter_format", p_start, t["format_ns"]))
|
|
510
|
+
lm_start = p_start + t["format_ns"]
|
|
511
|
+
lm_tokens = t["lm_tokens"]
|
|
512
|
+
completion = lm_tokens // 10
|
|
513
|
+
prompt = lm_tokens - completion
|
|
514
|
+
total_tokens += lm_tokens
|
|
515
|
+
spans.append(
|
|
516
|
+
{
|
|
517
|
+
"span_id": str(uuid.uuid4()),
|
|
518
|
+
"trace_id": trace_id,
|
|
519
|
+
"parent_span_id": predict_id,
|
|
520
|
+
"name": "LM.__call__",
|
|
521
|
+
"span_type": "lm_call",
|
|
522
|
+
"start_time": at(lm_start),
|
|
523
|
+
"end_time": at(lm_start + t["lm_ns"]),
|
|
524
|
+
"duration_ns": t["lm_ns"],
|
|
525
|
+
"status": "ok",
|
|
526
|
+
"model_name": model,
|
|
527
|
+
"model_provider": "bedrock",
|
|
528
|
+
"prompt_tokens": prompt,
|
|
529
|
+
"completion_tokens": completion,
|
|
530
|
+
"total_tokens": lm_tokens,
|
|
531
|
+
}
|
|
532
|
+
)
|
|
533
|
+
spans.append(
|
|
534
|
+
adapter_span(predict_id, "ChatAdapter.parse", "adapter_parse", p_end - t["parse_ns"], t["parse_ns"])
|
|
535
|
+
)
|
|
536
|
+
cursor = p_end + gap_ns
|
|
537
|
+
|
|
538
|
+
# No end_time / duration_ns at the trace level either — a live trace has not
|
|
539
|
+
# closed. status "in_progress" so serve renders the running notebook.
|
|
540
|
+
return {
|
|
541
|
+
"trace_id": trace_id,
|
|
542
|
+
"start_time": at(0),
|
|
543
|
+
"status": "in_progress",
|
|
544
|
+
"program_name": "RLM",
|
|
545
|
+
"program_version": "0.1.0",
|
|
546
|
+
"program_signature_hash": fx["program_signature_hash"],
|
|
547
|
+
"signature_name": fx["signature_name"],
|
|
548
|
+
"signature_input_fields": fx["input_fields"],
|
|
549
|
+
"signature_output_fields": fx["output_fields"],
|
|
550
|
+
"adapter_type": "ChatAdapter",
|
|
551
|
+
"request_preview": "Colorado 2026 U.S. House elections (Ballotpedia markdown)",
|
|
552
|
+
"response_preview": "",
|
|
553
|
+
"project_tags": _SEED_TAGS + [_GOLDEN_PATH, "notebook-demo", "in-progress"],
|
|
388
554
|
"total_tokens": total_tokens,
|
|
389
555
|
"span_count": len(spans),
|
|
390
556
|
"spans": spans,
|
|
@@ -514,6 +680,85 @@ def _stamp_deployments(traces: list[dict], dep_for) -> None:
|
|
|
514
680
|
trace["deployment_id"] = dep
|
|
515
681
|
|
|
516
682
|
|
|
683
|
+
# The review app's bootstrap admin (serve devEasySeedUser): admin@review.local.
|
|
684
|
+
# Its password is the review release's fixed default — release_serve_review.py's
|
|
685
|
+
# DEFAULT_SEED_ADMIN_PASSWORD, which the review flow never overrides — so the
|
|
686
|
+
# default below matches the deployed password with no workflow env needed.
|
|
687
|
+
# CMPND_SERVE_SEED_ADMIN_PASSWORD stays an override hook for the day that
|
|
688
|
+
# password is ever customized.
|
|
689
|
+
_ADMIN_EMAIL = "admin@review.local"
|
|
690
|
+
_ADMIN_PASSWORD_DEFAULT = "review-app-admin-password"
|
|
691
|
+
_CSRF_INPUT_RE = re.compile(r'name="csrf_token"\s+value="([^"]+)"')
|
|
692
|
+
_CSRF_META_RE = re.compile(r'name="csrf-token"\s+content="([^"]+)"')
|
|
693
|
+
|
|
694
|
+
|
|
695
|
+
def _display_program_hash(signed_hash: int) -> str:
|
|
696
|
+
"""The 16-hex display form of a signed-int64 program_signature_hash — the
|
|
697
|
+
shape the /ui/programs/{hash}/pin route parses (ids.FormatProgramHashInt64
|
|
698
|
+
in serve: fmt.Sprintf("%016x", uint64(n)))."""
|
|
699
|
+
return format(signed_hash & 0xFFFFFFFFFFFFFFFF, "016x")
|
|
700
|
+
|
|
701
|
+
|
|
702
|
+
def _scrape_csrf(html: str) -> str | None:
|
|
703
|
+
"""Pull a CSRF token from a serve HTML page — the hidden `csrf_token` form
|
|
704
|
+
input, or the `<meta name="csrf-token">` fallback (both carry the per-session
|
|
705
|
+
token)."""
|
|
706
|
+
m = _CSRF_INPUT_RE.search(html) or _CSRF_META_RE.search(html)
|
|
707
|
+
return m.group(1) if m else None
|
|
708
|
+
|
|
709
|
+
|
|
710
|
+
def _pin_committee_program(endpoint: str, display_hash: str) -> None:
|
|
711
|
+
"""Pin the completed committee program for the demo admin (CMP-634).
|
|
712
|
+
|
|
713
|
+
Pins are per-user web state: `program_pins` FKs the ingested `programs` row
|
|
714
|
+
and serve's Pin() is reachable ONLY through the session+CSRF web route
|
|
715
|
+
POST /ui/programs/{hash}/pin — the API key (bearer, no cookies) can't hit it.
|
|
716
|
+
So we drive the browser flow with a cookie jar: log in as the bootstrap admin,
|
|
717
|
+
harvest a fresh CSRF token from the program page, then POST the pin. Wholly
|
|
718
|
+
non-gating — any failure logs and returns, like the rest of the seed.
|
|
719
|
+
|
|
720
|
+
Not exercisable against local dev (empty-password admin); it runs for real in
|
|
721
|
+
the review app, whose admin password is _ADMIN_PASSWORD_DEFAULT (the review
|
|
722
|
+
release's fixed default).
|
|
723
|
+
"""
|
|
724
|
+
password = os.getenv("CMPND_SERVE_SEED_ADMIN_PASSWORD") or _ADMIN_PASSWORD_DEFAULT
|
|
725
|
+
jar = http.cookiejar.CookieJar()
|
|
726
|
+
opener = urllib.request.build_opener(urllib.request.HTTPCookieProcessor(jar))
|
|
727
|
+
|
|
728
|
+
def get(path: str) -> str:
|
|
729
|
+
req = urllib.request.Request(endpoint + path, method="GET")
|
|
730
|
+
with opener.open(req, timeout=30) as resp: # noqa: S310 - fixed endpoint from env
|
|
731
|
+
return resp.read().decode("utf-8", "replace")
|
|
732
|
+
|
|
733
|
+
def post(path: str, fields: dict[str, str]) -> int:
|
|
734
|
+
data = urllib.parse.urlencode(fields).encode("utf-8")
|
|
735
|
+
req = urllib.request.Request(
|
|
736
|
+
endpoint + path,
|
|
737
|
+
data=data,
|
|
738
|
+
method="POST",
|
|
739
|
+
headers={"Content-Type": "application/x-www-form-urlencoded"},
|
|
740
|
+
)
|
|
741
|
+
with opener.open(req, timeout=30) as resp: # noqa: S310 - fixed endpoint from env
|
|
742
|
+
return resp.status
|
|
743
|
+
|
|
744
|
+
# 1) GET the login page → session cookie + a CSRF token to submit with login.
|
|
745
|
+
login_csrf = _scrape_csrf(get("/auth/login"))
|
|
746
|
+
if not login_csrf:
|
|
747
|
+
log.warning("pin: no CSRF token on /auth/login; skipping pin")
|
|
748
|
+
return
|
|
749
|
+
# 2) Log in. serve rotates the CSRF token on login, so the token above is
|
|
750
|
+
# single-use for this POST only.
|
|
751
|
+
post("/auth/login", {"email": _ADMIN_EMAIL, "password": password, "csrf_token": login_csrf})
|
|
752
|
+
# 3) Harvest a FRESH CSRF token from the (now authenticated) program page.
|
|
753
|
+
pin_csrf = _scrape_csrf(get(f"/ui/programs/{display_hash}"))
|
|
754
|
+
if not pin_csrf:
|
|
755
|
+
log.warning("pin: no CSRF token on program page (login may have failed); skipping pin")
|
|
756
|
+
return
|
|
757
|
+
# 4) POST the pin.
|
|
758
|
+
status = post(f"/ui/programs/{display_hash}/pin", {"csrf_token": pin_csrf})
|
|
759
|
+
log.info("pinned committee program %s (status %s)", display_hash, status)
|
|
760
|
+
|
|
761
|
+
|
|
517
762
|
def main() -> int:
|
|
518
763
|
endpoint = (os.getenv("CMPND_E2E_ENDPOINT") or "").rstrip("/")
|
|
519
764
|
api_key = os.getenv("CMPND_E2E_API_KEY") or ""
|
|
@@ -579,6 +824,20 @@ def main() -> int:
|
|
|
579
824
|
n_spans += 1
|
|
580
825
|
log.info("seeded %d traces, %d span-batches, %d stat rows -> %s", n_traces, n_spans, n_stats, new_op)
|
|
581
826
|
|
|
827
|
+
# 2b) Pin the completed committee program for the demo admin (CMP-634) so
|
|
828
|
+
# the sidebar shows a pinned program. Runs now that the trace ingest has
|
|
829
|
+
# upserted the programs row the pin FKs. Self-contained try/except so the
|
|
830
|
+
# login+web-route flow (the only path to a per-user pin) can never break
|
|
831
|
+
# the seed.
|
|
832
|
+
try:
|
|
833
|
+
committee_hash = opt["payload"].get("program_signature_hash")
|
|
834
|
+
if committee_hash is not None:
|
|
835
|
+
_pin_committee_program(endpoint, _display_program_hash(int(committee_hash)))
|
|
836
|
+
else:
|
|
837
|
+
log.warning("pin: optimization payload has no program_signature_hash; skipping pin")
|
|
838
|
+
except Exception as exc: # noqa: BLE001 - non-gating, like the rest of the seed
|
|
839
|
+
log.warning("pin committee program failed (non-gating): %s", exc)
|
|
840
|
+
|
|
582
841
|
# 3) Plant the interrupted twin of the completed run — a rich `running`
|
|
583
842
|
# row frozen right after iteration 2 (real iterations/candidates/scores
|
|
584
843
|
# + its own linked traces), sharing the completed run's `log_dir` so the
|
|
@@ -651,6 +910,14 @@ def main() -> int:
|
|
|
651
910
|
code, _ = _post(endpoint, "/api/v1/traces/batch", notebook_batch, api_key)
|
|
652
911
|
log.info("seeded RLM notebook trace (status %s)", code)
|
|
653
912
|
|
|
913
|
+
# 4a) Plant a SECOND DistrictCounter RLM trace, frozen mid-flight as an
|
|
914
|
+
# in_progress live notebook (CMP-634) so the demo shows the live-streaming
|
|
915
|
+
# autofill next to the completed run above. No end_time, partial trajectory,
|
|
916
|
+
# status in_progress → serve renders the "running" notebook. Left unstamped
|
|
917
|
+
# (a live run predates any deployment link).
|
|
918
|
+
code, _ = _post(endpoint, "/api/v1/traces/batch", [_notebook_trace_in_progress(now)], api_key)
|
|
919
|
+
log.info("seeded in-progress RLM notebook trace (status %s)", code)
|
|
920
|
+
|
|
654
921
|
# 4b) Plant the two CMP-187 tool-span demo traces so every review app
|
|
655
922
|
# shows RLM tool spans (SourceCategorizer, with a vision sub-LLM nested
|
|
656
923
|
# under Tool.inspect_page) and the function-tool vs module-tool
|
|
@@ -683,6 +950,22 @@ def main() -> int:
|
|
|
683
950
|
)
|
|
684
951
|
else:
|
|
685
952
|
log.warning("skipping ToolKindDemo seed: fixture missing at %s", _TOOL_KIND_DEMO_FIXTURE)
|
|
953
|
+
# A dspy.ReAct(ExtractCommittee) trace (CMP-634): reasons, calls a
|
|
954
|
+
# `lookup_disclaimer` tool to pull the "Paid for by" line, then extracts
|
|
955
|
+
# the committee — the tool-span ReAct exemplar the golden-path demo lacked.
|
|
956
|
+
if _REACT_COMMITTEE_FIXTURE.exists():
|
|
957
|
+
tool_demos.append(
|
|
958
|
+
_replay_flat_fixture(
|
|
959
|
+
_REACT_COMMITTEE_FIXTURE,
|
|
960
|
+
now,
|
|
961
|
+
"ReAct",
|
|
962
|
+
_SEED_TAGS + [_GOLDEN_PATH, "react-demo"],
|
|
963
|
+
"committee fundraising email (ReAct + lookup_disclaimer tool)",
|
|
964
|
+
"NATIONAL REPUBLICAN SENATORIAL COMMITTEE",
|
|
965
|
+
)
|
|
966
|
+
)
|
|
967
|
+
else:
|
|
968
|
+
log.warning("skipping ReAct committee seed: fixture missing at %s", _REACT_COMMITTEE_FIXTURE)
|
|
686
969
|
if tool_demos:
|
|
687
970
|
_stamp_deployments(tool_demos, dep_for)
|
|
688
971
|
code, _ = _post(endpoint, "/api/v1/traces/batch", tool_demos, api_key)
|
|
@@ -47,10 +47,22 @@ _DATA_DIR = pathlib.Path(__file__).resolve().parent / "data"
|
|
|
47
47
|
# Bedrock's OpenAI-compatible surface, where they are `404 model_not_found` (they live on
|
|
48
48
|
# the native Converse surface litellm translates to), so a run against that plane, or
|
|
49
49
|
# against any other provider, has to be able to say so without editing this file.
|
|
50
|
-
|
|
51
|
-
|
|
50
|
+
#
|
|
51
|
+
# The DEPLOYED_ variables, because that is what every model here is for: this module's
|
|
52
|
+
# programs are built to be deployed and executed by a data plane, so the plane's provider
|
|
53
|
+
# constraint is theirs. The un-prefixed `PLATFORM_*_MODEL` names configure LMs that a test
|
|
54
|
+
# process calls directly, and reading those here is what made a wasm-plane run also rewire
|
|
55
|
+
# tests that never touch a plane. `_provider` (which this module cannot import — it
|
|
56
|
+
# predates the stress suite's resolver and is shared with a lane that has no wasm plane)
|
|
57
|
+
# carries the full note.
|
|
58
|
+
STUDENT_1B = (
|
|
59
|
+
os.getenv("PLATFORM_DEPLOYED_STRESS_MODEL") or "bedrock/us.amazon.nova-micro-v1:0"
|
|
60
|
+
) # tiny/cheap; no lift assertion
|
|
61
|
+
STUDENT_3B = (
|
|
62
|
+
os.getenv("PLATFORM_DEPLOYED_STUDENT_MODEL") or "bedrock/google.gemma-3-4b-it"
|
|
63
|
+
) # weak-with-headroom; gepa student
|
|
52
64
|
REFLECTION_MODEL = (
|
|
53
|
-
os.getenv("
|
|
65
|
+
os.getenv("PLATFORM_DEPLOYED_REFLECTION_MODEL") or "bedrock/us.meta.llama3-3-70b-instruct-v1:0"
|
|
54
66
|
) # gepa reflection (resolved server-side)
|
|
55
67
|
|
|
56
68
|
|
|
@@ -89,8 +101,11 @@ def lm_api_base() -> str | None:
|
|
|
89
101
|
``lm_api_host``, which seeds the org's provider key, which is what the plane later
|
|
90
102
|
resolves into ``inference.student.api_base``. Without it a run there fails on the
|
|
91
103
|
broker's own "no api_base" refusal.
|
|
104
|
+
|
|
105
|
+
``PLATFORM_DEPLOYED_LM_API_BASE``, matching the model ids above: this base is declared
|
|
106
|
+
on a program the PLANE runs, so it must not also redirect a test's own litellm calls.
|
|
92
107
|
"""
|
|
93
|
-
return os.getenv("
|
|
108
|
+
return os.getenv("PLATFORM_DEPLOYED_LM_API_BASE") or None
|
|
94
109
|
|
|
95
110
|
|
|
96
111
|
def _lm_kwargs() -> dict[str, Any]:
|