cmpnd 0.6.0__tar.gz → 0.6.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {cmpnd-0.6.0 → cmpnd-0.6.2}/CLAUDE.md +17 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/PKG-INFO +1 -1
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/__init__.py +2 -2
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/_gepa_patch.py +3 -3
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/_program_patch.py +1 -1
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/_rlm_patch.py +1 -1
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/callback.py +9 -9
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/cli/__init__.py +2 -2
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/cli/commands/admin.py +1 -1
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/cli/commands/org.py +3 -3
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/cli/commands/top.py +3 -3
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/cli/shell.py +2 -2
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/cli/trace_render.py +2 -2
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/configuration.py +2 -2
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/context.py +1 -1
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/deployment.py +5 -5
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/eval_handler.py +2 -2
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/execution.py +42 -1
- cmpnd-0.6.2/cmpnd/exporter.py +108 -0
- cmpnd-0.6.0/cmpnd/exporter.py → cmpnd-0.6.2/cmpnd/exporter_http.py +26 -64
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/helpers.py +1 -1
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/identity.py +15 -6
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/ir/encode.py +2 -4
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/ir/hash.py +2 -4
- cmpnd-0.6.2/cmpnd/ir/xxh64.py +131 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/models.py +11 -11
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/optimization.py +2 -2
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/optimizers/gepa.py +1 -1
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/optimizers/gepa_callback.py +7 -7
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/optimizers/resume.py +1 -1
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/optimizers/tracker.py +10 -10
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/packaging.py +11 -11
- {cmpnd-0.6.0 → cmpnd-0.6.2}/docs/sdk-instrumentation.md +18 -9
- {cmpnd-0.6.0 → cmpnd-0.6.2}/pyproject.toml +7 -1
- {cmpnd-0.6.0 → cmpnd-0.6.2}/scripts/capture_review_run.py +2 -2
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/e2e/committee_task.py +45 -8
- cmpnd-0.6.2/tests/stress/_provider.py +91 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/stress/test_deploy_programs.py +8 -11
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/stress/test_multi_client_stress.py +9 -12
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/stress/test_optimize_client_committee.py +6 -11
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/stress/test_optimize_loop.py +11 -14
- cmpnd-0.6.2/tests/stress/test_optimize_loop_wasm.py +406 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/stress/test_optimize_stress.py +8 -11
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/stress/test_real_provider_twins.py +8 -11
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_identity.py +34 -0
- cmpnd-0.6.2/tests/test_import_weight.py +138 -0
- cmpnd-0.6.2/tests/test_xxh64.py +101 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/uv.lock +1 -1
- {cmpnd-0.6.0 → cmpnd-0.6.2}/.gitignore +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/README.md +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/cli/api_client.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/cli/commands/__init__.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/cli/commands/auth.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/cli/commands/datasets.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/cli/commands/deployments.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/cli/commands/evals.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/cli/commands/login.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/cli/commands/optimizations.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/cli/commands/programs.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/cli/commands/traces.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/cli/credentials.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/cli/verbs.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/dataset_sync.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/datasets.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/decorators.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/ir/__init__.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/ir/decode.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/ir/program.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/ir/reflection.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/ir/types.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/cmpnd/optimizers/__init__.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/docs/2026-04-02-code-review.md +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/docs/plans/2026-04-05-train-val-dataset-capture-design.md +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/docs/plans/2026-04-05-train-val-dataset-capture.md +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/examples/local_ollama.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/scripts/fixtures/review_run.json.gz +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/scripts/gepa_resume_bug.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/scripts/seed_review_run.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/stubs/dspy/__init__.pyi +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/stubs/dspy/primitives/__init__.pyi +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/stubs/dspy/primitives/prediction.pyi +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/stubs/dspy/utils/__init__.pyi +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/stubs/dspy/utils/callback.pyi +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/__init__.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/_exporter_helpers.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/_identity_helpers.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/e2e/__init__.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/e2e/conftest.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/e2e/data/committee_train.ndjson.gz +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/e2e/data/committee_val.ndjson.gz +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/e2e/fake_lm.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/e2e/helpers.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/e2e/test_capture_prompts.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/e2e/test_composite_module.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/e2e/test_dataset_auto_creation.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/e2e/test_deploy_execute.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/e2e/test_deploy_programs.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/e2e/test_edge_cases.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/e2e/test_example_hash.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/e2e/test_gepa_optimization.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/e2e/test_lm_accuracy.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/e2e/test_module_tracing.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/e2e/test_optimization_lineage.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/e2e/test_optimization_metric_identity.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/e2e/test_optimize_failures.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/e2e/test_optimize_loop.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/e2e/test_optimize_review_committee.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/e2e/test_optimize_server_owned_completion.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/e2e/test_parallel_export.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/e2e/test_project_attribution.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/e2e/test_react_tool_spans.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/e2e/test_save_load.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/e2e/test_trace_structure.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/e2e/test_user_attribution.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/identity_vectors/01_predict_qa.json +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/identity_vectors/02_predict_renamed_field.json +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/identity_vectors/03_chain_of_thought_qa.json +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/identity_vectors/04_predict_richer_types.json +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/identity_vectors/05_composite_two_predicts.json +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/identity_vectors/06_multichaincomparison.json +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/identity_vectors/07_program_of_thought.json +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/identity_vectors/08_opaque_module.json +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/identity_vectors/09_predict_pydantic_field.json +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/identity_vectors/10_react.json +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/identity_vectors/11_refine.json +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/identity_vectors/12_retrieve.json +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/identity_vectors/13_knn.json +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/identity_vectors/14_parallel.json +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/identity_vectors/README.md +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/integration/__init__.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/integration/conftest.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/integration/test_cli_deploy_smoketest.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/integration/test_logs_renders_traces.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/integration/test_optimizing_progress.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/integration/test_unhappy_path.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/parity/__init__.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/parity/conftest.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/parity/seed_parity.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/parity/test_auth.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/parity/test_datasets.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/parity/test_deployments.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/parity/test_evals.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/parity/test_health.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/parity/test_modules.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/parity/test_optimizations.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/parity/test_programs.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/parity/test_signatures.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/parity/test_trace_streaming.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/parity/test_traces.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/stress/__init__.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/stress/_reliability.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/stress/mint_bedrock_token.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/stress/test_reliability.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/stress/test_reliability_classify.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_callback.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_cli/__init__.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_cli/conftest.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_cli/test_admin.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_cli/test_api_client.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_cli/test_auth.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_cli/test_credentials.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_cli/test_datasets.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_cli/test_deployments.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_cli/test_evals.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_cli/test_login.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_cli/test_main.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_cli/test_optimizations.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_cli/test_org.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_cli/test_parity.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_cli/test_programs.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_cli/test_shell.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_cli/test_top.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_cli/test_traces.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_clustering/__init__.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_clustering/conftest.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_clustering/test_edge_cases.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_clustering/test_hash_consistency.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_clustering/test_type_conversion.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_cmp551_grouping_identity.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_configuration.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_context.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_context_propagation.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_dataset_polling.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_dataset_sync.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_datasets.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_decorators.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_deploy.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_eval.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_eval_start.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_exception_paths.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_execute.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_exporter.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_exporter_real.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_fake_lm.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_gepa_callback.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_gepa_integration.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_gepa_resume.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_integration.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_ir_encoder_properties.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_ir_golden_vectors.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_ir_roundtrip.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_models.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_optimization_progress.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_optimization_start.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_optimize.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_optimize_lift.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_optimize_terminal_failure.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_packaging.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_packaging_callable_source.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_packaging_program_source.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_program_lineage_persist.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_project_payload.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_reflector_properties.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_reflector_smoke.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_repro_reactv2_bugs.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_rlm_patch.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_schemas.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_seed_review_run.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_stats_export.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_trace_render.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_tracked_gepa.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_tracker_enrich.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/tests/test_truncation.py +0 -0
- {cmpnd-0.6.0 → cmpnd-0.6.2}/todo.md +0 -0
|
@@ -2,6 +2,23 @@
|
|
|
2
2
|
|
|
3
3
|
Part of the CMPND monorepo. See `../CLAUDE.md` for cross-cutting rules (hash architecture, null handling, testing philosophy).
|
|
4
4
|
|
|
5
|
+
## This code is released publicly — no private references
|
|
6
|
+
|
|
7
|
+
Unlike the rest of the monorepo (private), the `cmpnd` package ships to PyPI:
|
|
8
|
+
`python-sdk/cmpnd/**` is **public source**. Anything written there is
|
|
9
|
+
world-readable in the published wheel. So **never put private information in the
|
|
10
|
+
shipped package** — no internal issue-tracker IDs (`CMP-…`), internal URLs or
|
|
11
|
+
dashboards, customer/org names, infra hostnames, or secrets — and that includes
|
|
12
|
+
comments, docstrings, `--help`/CLI strings, error messages, and log lines, not
|
|
13
|
+
just code. Reference an internal ticket in the **commit message or PR
|
|
14
|
+
description** (those live in the private repo), never in the source that ships.
|
|
15
|
+
|
|
16
|
+
This applies only to the shipped package (`python-sdk/cmpnd/`). Tests
|
|
17
|
+
(`python-sdk/tests/`), `docs/`, and `docs/plans/` are not published in the
|
|
18
|
+
wheel, but prefer keeping issue IDs out of docstrings you might later move into
|
|
19
|
+
the package. When you cite prior work in shipped code, describe the behavior,
|
|
20
|
+
not the ticket.
|
|
21
|
+
|
|
5
22
|
## Documentation discipline (docs are as-built, updated in the same commit)
|
|
6
23
|
|
|
7
24
|
This mirrors the practice that has kept `serve/docs/` in sync with the Go
|
|
@@ -131,13 +131,13 @@ def auto_instrument() -> CmpndCallback:
|
|
|
131
131
|
|
|
132
132
|
# Persist the compiled-from lineage tag across state-only save/load so a
|
|
133
133
|
# continued optimization still stamps compiled_from_run_id after a disk
|
|
134
|
-
# reload / process restart
|
|
134
|
+
# reload / process restart.
|
|
135
135
|
from . import _program_patch
|
|
136
136
|
|
|
137
137
|
_program_patch.patch_program_state()
|
|
138
138
|
|
|
139
139
|
# Emit tool spans for dspy.RLM tool calls, which otherwise bypass
|
|
140
|
-
# dspy.Tool.__call__'s callbacks and produce no spans
|
|
140
|
+
# dspy.Tool.__call__'s callbacks and produce no spans.
|
|
141
141
|
from . import _rlm_patch
|
|
142
142
|
|
|
143
143
|
_rlm_patch.patch_rlm_tools()
|
|
@@ -84,7 +84,7 @@ def patch_gepa() -> None:
|
|
|
84
84
|
|
|
85
85
|
# Build the config dict from the GEPA instance, then hand the live
|
|
86
86
|
# program to the shared enrichment helper (one definition of the
|
|
87
|
-
# optimization_run payload shape, reused by the deploy path
|
|
87
|
+
# optimization_run payload shape, reused by the deploy path).
|
|
88
88
|
config = {}
|
|
89
89
|
for key in ["auto", "max_metric_calls", "max_full_evals", "seed", "log_dir"]:
|
|
90
90
|
val = getattr(self, key, None)
|
|
@@ -97,7 +97,7 @@ def patch_gepa() -> None:
|
|
|
97
97
|
except Exception:
|
|
98
98
|
pass
|
|
99
99
|
|
|
100
|
-
# Restore the up-front budget estimate
|
|
100
|
+
# Restore the up-front budget estimate: for an `auto`
|
|
101
101
|
# preset `self.max_metric_calls` is None until GEPA's first budget
|
|
102
102
|
# event, so the initial "running" record shipped 0 and the progress
|
|
103
103
|
# bar / ETA had no denominator. Estimate now instead.
|
|
@@ -160,7 +160,7 @@ def patch_gepa() -> None:
|
|
|
160
160
|
except Exception as e:
|
|
161
161
|
if callback is not None: # pyright: ignore[reportUnnecessaryComparison]
|
|
162
162
|
# Flush a terminal 'failed', not just in-memory state — else the
|
|
163
|
-
# running row start() sent lingers forever
|
|
163
|
+
# running row start() sent lingers forever.
|
|
164
164
|
callback._tracker.finalize_error(e)
|
|
165
165
|
raise
|
|
166
166
|
|
|
@@ -17,7 +17,7 @@ the state-only path, which is the one DSPy steers users toward.)
|
|
|
17
17
|
|
|
18
18
|
DSPy exposes no extension point for out-of-band instance state (no
|
|
19
19
|
``get_extra_state`` / attribute allowlist; ``metadata`` is DSPy-owned). Rather
|
|
20
|
-
than change DSPy upstream
|
|
20
|
+
than change DSPy upstream, ``patch_program_state()`` wraps
|
|
21
21
|
``dump_state`` / ``load_state`` on the two classes that define them — the
|
|
22
22
|
composite ``Module`` path and the bare ``Predict`` path — to round-trip the tag
|
|
23
23
|
under a namespaced carrier key. Installed lazily from
|
|
@@ -71,7 +71,7 @@ class _CallbackBase:
|
|
|
71
71
|
self._capture_inputs = (
|
|
72
72
|
capture_inputs if capture_inputs is not None else (config.capture_inputs if config else True)
|
|
73
73
|
)
|
|
74
|
-
# Emit an in_progress start for every root trace, not just RLM
|
|
74
|
+
# Emit an in_progress start for every root trace, not just RLM.
|
|
75
75
|
self._stream_trace_starts = (
|
|
76
76
|
stream_trace_starts if stream_trace_starts is not None else (config.stream_trace_starts if config else True)
|
|
77
77
|
)
|
|
@@ -117,7 +117,7 @@ class _CallbackBase:
|
|
|
117
117
|
self._eval_handler = eval_handler.EvalHandler(self)
|
|
118
118
|
|
|
119
119
|
# Active optimization run ID - stored on instance for same thread-safety reason.
|
|
120
|
-
# Prefixed text id (`op_…`), minted via cmpnd.identity
|
|
120
|
+
# Prefixed text id (`op_…`), minted via cmpnd.identity.
|
|
121
121
|
self._active_optimization_run_id: str | None = None
|
|
122
122
|
|
|
123
123
|
# Token usage from LM calls without a trace context (e.g., GEPA reflection calls)
|
|
@@ -188,7 +188,7 @@ class _ModuleMixin(_CallbackBase):
|
|
|
188
188
|
context.set_current_trace(trace)
|
|
189
189
|
|
|
190
190
|
# Carry the root module's type so serve can classify the trace at
|
|
191
|
-
# start
|
|
191
|
+
# start — before any span lands — instead of inferring it
|
|
192
192
|
# from span shape later.
|
|
193
193
|
trace.root_span_type = span_type.value
|
|
194
194
|
# RLM traces use streaming export (per-span); everything else
|
|
@@ -197,7 +197,7 @@ class _ModuleMixin(_CallbackBase):
|
|
|
197
197
|
trace.streaming = True
|
|
198
198
|
self._streaming_traces[str(trace.trace_id)] = trace.streaming
|
|
199
199
|
|
|
200
|
-
# Emit an in_progress start row for EVERY root trace
|
|
200
|
+
# Emit an in_progress start row for EVERY root trace, not
|
|
201
201
|
# just streaming/RLM ones, so a non-RLM program is visible the moment
|
|
202
202
|
# it starts (`traces search --status in_progress`, a lookupable id)
|
|
203
203
|
# instead of only appearing at finalize. `trace.streaming` still
|
|
@@ -251,7 +251,7 @@ class _ModuleMixin(_CallbackBase):
|
|
|
251
251
|
# Surface the previous turn's code output on THIS (turn N+1)
|
|
252
252
|
# span so a streaming reader can fill in turn N's notebook
|
|
253
253
|
# output mid-run, rather than waiting for the whole trajectory
|
|
254
|
-
# on the RLM parent span at finalize
|
|
254
|
+
# on the RLM parent span at finalize. The prior
|
|
255
255
|
# turn's span has already ended + been exported, so it rides
|
|
256
256
|
# here instead; the RLM's final turn (no successor) still
|
|
257
257
|
# resolves at finalize. Capped like an output field so a large
|
|
@@ -433,7 +433,7 @@ class _ModuleMixin(_CallbackBase):
|
|
|
433
433
|
trace._batch_spans = trace_spans
|
|
434
434
|
exp.export_trace(trace)
|
|
435
435
|
|
|
436
|
-
# Correlate the returned Prediction to its trace
|
|
436
|
+
# Correlate the returned Prediction to its trace.
|
|
437
437
|
# Stored under a leading underscore so it stays a plain instance
|
|
438
438
|
# attribute — a bare name would route into dspy.Prediction's field
|
|
439
439
|
# store and surface as a phantom output field. Read it back via the
|
|
@@ -785,7 +785,7 @@ class _LMMixin(_CallbackBase):
|
|
|
785
785
|
# new instructions, between evaluations) and metric (an LLM-as-judge /
|
|
786
786
|
# semantic metric scoring outputs, during evaluation). Rather than drop
|
|
787
787
|
# them, wrap each in its own trace linked to the optimization run so it
|
|
788
|
-
# shows up alongside the student traces
|
|
788
|
+
# shows up alongside the student traces. serve's
|
|
789
789
|
# TracedTokensForRun then accounts for their cost from the span, so these
|
|
790
790
|
# must NOT also be added to _orphaned_lm_usage / untraced_lm_usage — that
|
|
791
791
|
# would double-count them in the optimization total.
|
|
@@ -821,7 +821,7 @@ class _LMMixin(_CallbackBase):
|
|
|
821
821
|
is running — instead of assuming every untraced call is reflection (that
|
|
822
822
|
would mislabel judge calls). The trace is named with the umbrella "GEPA";
|
|
823
823
|
the specific model stays on the span's model_name, and `gepa.phase`
|
|
824
|
-
carries the distinction
|
|
824
|
+
carries the distinction. Each becomes a standalone one-span
|
|
825
825
|
trace so the prompt, response, tokens and cost are visible alongside the
|
|
826
826
|
student traces instead of being reduced to an aggregate token count.
|
|
827
827
|
"""
|
|
@@ -842,7 +842,7 @@ class _LMMixin(_CallbackBase):
|
|
|
842
842
|
# context stack, so on_lm_start stamped this span with a parent_span_id for
|
|
843
843
|
# a span that isn't in this standalone one-span trace. This span IS the
|
|
844
844
|
# trace's root, so force it parentless and pin it to this trace so tree/root
|
|
845
|
-
# reconstruction works
|
|
845
|
+
# reconstruction works.
|
|
846
846
|
span.parent_span_id = None
|
|
847
847
|
span.trace_id = trace.trace_id
|
|
848
848
|
trace.signature_name = span.model_name or phase
|
|
@@ -87,7 +87,7 @@ def main() -> None:
|
|
|
87
87
|
auth.register(subparsers)
|
|
88
88
|
login.register(subparsers)
|
|
89
89
|
|
|
90
|
-
# Org membership (owner-facing approval queue
|
|
90
|
+
# Org membership (owner-facing approval queue). Always
|
|
91
91
|
# registered — the server gates each action on org ownership, so a
|
|
92
92
|
# non-owner just gets a 403, not a hidden command.
|
|
93
93
|
org.register(subparsers)
|
|
@@ -99,7 +99,7 @@ def main() -> None:
|
|
|
99
99
|
if os.environ.get("CMPND_CLI_ADMIN") == "1":
|
|
100
100
|
admin.register(subparsers)
|
|
101
101
|
|
|
102
|
-
#
|
|
102
|
+
# Resource subcommands are dev-only until productized.
|
|
103
103
|
# Set CMPND_CLI_DEV=1 to expose programs, traces, optimizations,
|
|
104
104
|
# evals, datasets, and deployments in --help.
|
|
105
105
|
if os.environ.get("CMPND_CLI_DEV") == "1":
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
"""Platform-operator commands, hidden behind CMPND_CLI_ADMIN=1.
|
|
2
2
|
|
|
3
3
|
`cmpnd admin org create|set-owner|list|get|set-limit|set-github-login|elect-owner`
|
|
4
|
-
is the closed-beta org bootstrap and steward surface
|
|
4
|
+
is the closed-beta org bootstrap and steward surface:
|
|
5
5
|
stand up a shared org with its GitHub auto-join key, anoint its first owner,
|
|
6
6
|
adjust its member cap or GitHub link, and grant ownership from the election
|
|
7
7
|
queue. `cmpnd admin user …` is the operator's user-lifecycle surface (list,
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
"""Org membership + ownership commands (customer-facing).
|
|
2
2
|
|
|
3
|
-
`cmpnd org members …` is the owner's delegated membership surface
|
|
4
|
-
|
|
3
|
+
`cmpnd org members …` is the owner's delegated membership surface: list
|
|
4
|
+
the GitHub users who auto-joined and are awaiting approval and
|
|
5
5
|
admit/decline them (the pull direction), and issue/list/revoke single-use
|
|
6
6
|
invite links for teammates you want to add (the push direction — `invite`
|
|
7
7
|
prints a URL you share out-of-band; the invitee joins as a member when they
|
|
@@ -15,7 +15,7 @@ hit /api/v1/org/ownership-requests and are self-scoped to the calling key's
|
|
|
15
15
|
user — no owner check (a platform admin grants ownership from the election
|
|
16
16
|
queue; the request is just recorded intent).
|
|
17
17
|
|
|
18
|
-
`cmpnd org keys list|create|revoke` is your own API-key surface
|
|
18
|
+
`cmpnd org keys list|create|revoke` is your own API-key surface:
|
|
19
19
|
list the keys you've minted, mint a new scoped key (the response carries the
|
|
20
20
|
one-time token — the server stores only its hash), and revoke one. These hit
|
|
21
21
|
/api/v1/org/keys and are self-scoped to the calling key's user — you only ever
|
|
@@ -113,8 +113,8 @@ def _hash_fallback(sig: Any) -> str:
|
|
|
113
113
|
is what someone would copy/paste. Names should be the common case once the
|
|
114
114
|
Program registry has caught up — this is just the fallback.
|
|
115
115
|
"""
|
|
116
|
-
# program_signature_hash now arrives as a decimal string on the wire
|
|
117
|
-
#
|
|
116
|
+
# program_signature_hash now arrives as a decimal string on the wire;
|
|
117
|
+
# int(sig) below handles both forms. "0" is the unset sentinel
|
|
118
118
|
# alongside numeric 0 / None / "".
|
|
119
119
|
if sig in (None, "", 0, "0"):
|
|
120
120
|
return "—"
|
|
@@ -379,7 +379,7 @@ def _render_status(d: dict[str, Any], endpoint: str) -> None:
|
|
|
379
379
|
print(f" Deploy prog: {deploy_program_id}")
|
|
380
380
|
print(f" Status: {_status(d.get('status', '—'))} ({age})")
|
|
381
381
|
print(f" Created: {d.get('created_at', '—')}")
|
|
382
|
-
# The caller's deploy(metadata=…) blob, if any
|
|
382
|
+
# The caller's deploy(metadata=…) blob, if any — the
|
|
383
383
|
# checkpoint/cursor that produced this program. Omitted when empty ({}).
|
|
384
384
|
metadata = d.get("metadata") or {}
|
|
385
385
|
if metadata:
|
|
@@ -92,7 +92,7 @@ def _logged_out_banner(endpoint: str) -> str:
|
|
|
92
92
|
f"{_c('cmpnd shell', 'bold')} — not signed in ({endpoint}).\n"
|
|
93
93
|
f" Set {_c('CMPND_API_KEY', 'bold')} (and optionally "
|
|
94
94
|
f"{_c('CMPND_ENDPOINT', 'bold')}), then restart.\n"
|
|
95
|
-
f" Browser-based {_c('cmpnd login', 'bold')} is on the way
|
|
95
|
+
f" Browser-based {_c('cmpnd login', 'bold')} is on the way.\n"
|
|
96
96
|
)
|
|
97
97
|
|
|
98
98
|
|
|
@@ -300,7 +300,7 @@ def _parse_signature_inputs(program_name: str | None) -> list[str]:
|
|
|
300
300
|
``run(body=None, **kwargs)`` form for those.
|
|
301
301
|
|
|
302
302
|
Stopgap: the deployment record only carries ``program_name`` as a
|
|
303
|
-
string today, not structured input/output fields.
|
|
303
|
+
string today, not structured input/output fields. A follow-up tracks
|
|
304
304
|
surfacing those on the serializer (sourced from what the deploy
|
|
305
305
|
service already produces for ``/openapi``). Once that lands, delete
|
|
306
306
|
this function and key off the structured fields instead — they'll
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
"""Render a trace's span forest as a tree or an ASCII timeline.
|
|
2
2
|
|
|
3
|
-
Consumed by ``cmpnd traces show <id
|
|
3
|
+
Consumed by ``cmpnd traces show <id>``. The raw
|
|
4
4
|
``GET /api/v1/traces/{id}`` response — ``{"trace": {...}, "spans": [...]}``
|
|
5
5
|
with flat, raw-column-named spans — is turned into something readable when the
|
|
6
6
|
caller opts into ``--tree`` / ``--timeline`` (``show`` is JSON by default, like
|
|
@@ -241,7 +241,7 @@ def span_line(span: dict[str, Any], *, color: bool) -> str:
|
|
|
241
241
|
"""One span rendered as a single line: '<type label> <name> · <duration> ·
|
|
242
242
|
<tokens> · <model>', mirroring serve's span-badge + span-name, then the same
|
|
243
243
|
duration string. Used both as a tree node body and as `traces follow`'s
|
|
244
|
-
per-span line
|
|
244
|
+
per-span line."""
|
|
245
245
|
# Two visually distinct segments: the span-type label carries the status
|
|
246
246
|
# hue (green/red/yellow), the span name is bold default-foreground. Bold on
|
|
247
247
|
# the terminal's own fg reads on any background; the two never share a color.
|
|
@@ -78,12 +78,12 @@ class Config:
|
|
|
78
78
|
max_queue_size: int = 10000
|
|
79
79
|
max_export_workers: int = 4
|
|
80
80
|
|
|
81
|
-
# Emit an in_progress trace_start for every root trace
|
|
81
|
+
# Emit an in_progress trace_start for every root trace, not just
|
|
82
82
|
# RLM, so non-RLM programs (Predict/ReAct/CoT/optimizer students) are
|
|
83
83
|
# visible while running instead of appearing only at finalize. The exporter
|
|
84
84
|
# coalesces start + batch in one flush, so a fast trace still pays a single
|
|
85
85
|
# insert. Set False to opt out of the extra start (traces then reappear only
|
|
86
|
-
# at finalize, the
|
|
86
|
+
# at finalize, the prior behavior).
|
|
87
87
|
stream_trace_starts: bool = True
|
|
88
88
|
|
|
89
89
|
# Feature flags
|
|
@@ -20,7 +20,7 @@ _span_stack: ContextVar[list[models.SpanBuilder] | None] = ContextVar("cmpnd_spa
|
|
|
20
20
|
_current_eval_run: ContextVar[models.EvalRunBuilder | None] = ContextVar("cmpnd_current_eval_run", default=None)
|
|
21
21
|
|
|
22
22
|
# Current optimization run context (set during optimization, used to link
|
|
23
|
-
# evals). Prefixed text id (`op_…`), minted via cmpnd.identity
|
|
23
|
+
# evals). Prefixed text id (`op_…`), minted via cmpnd.identity.
|
|
24
24
|
_current_optimization_run_id: ContextVar[str | None] = ContextVar("cmpnd_current_optimization_run_id", default=None)
|
|
25
25
|
|
|
26
26
|
|
|
@@ -20,7 +20,7 @@ from . import configuration, execution, identity, packaging
|
|
|
20
20
|
|
|
21
21
|
logger = logging.getLogger(__name__)
|
|
22
22
|
|
|
23
|
-
#
|
|
23
|
+
# The deploy metadata blob is a free-form JSON object committed with
|
|
24
24
|
# the deployment record (the checkpoint/cursor that produced the program). Cap
|
|
25
25
|
# its size — a checkpoint/cursor is a pointer, not a payload. Mirrors serve's
|
|
26
26
|
# maxMetadataBytes so the call-site error matches the server's 413.
|
|
@@ -127,7 +127,7 @@ def deploy(module, metric=None, timeout=300, metadata=None):
|
|
|
127
127
|
metadata: Optional free-form structured dict committed atomically with
|
|
128
128
|
the deployment record — somewhere to put the checkpoint/cursor that
|
|
129
129
|
produced this program, so the deployment is self-describing for a
|
|
130
|
-
continual-learning loop
|
|
130
|
+
continual-learning loop. Must be a JSON-serializable dict
|
|
131
131
|
with string keys, within 64 KiB. Read it back via the deployment
|
|
132
132
|
record (``cmpnd.status`` / the deployment GET API).
|
|
133
133
|
|
|
@@ -155,7 +155,7 @@ def deploy(module, metric=None, timeout=300, metadata=None):
|
|
|
155
155
|
# slot and reject an oversized deploy before signing it (Slice E).
|
|
156
156
|
"declared_content_length": len(zip_bytes),
|
|
157
157
|
}
|
|
158
|
-
#
|
|
158
|
+
# Carry the caller's metadata blob (validated at the call site) as a
|
|
159
159
|
# reserved top-level body key, committed with the deployment record server-side.
|
|
160
160
|
if metadata is not None:
|
|
161
161
|
_validate_deploy_metadata(metadata)
|
|
@@ -170,7 +170,7 @@ def deploy(module, metric=None, timeout=300, metadata=None):
|
|
|
170
170
|
|
|
171
171
|
# Forward LM credentials so the server can auto-create a Provider+ProviderKey.
|
|
172
172
|
#
|
|
173
|
-
#
|
|
173
|
+
# Credential policy: forward ONLY a key explicitly set on the LM
|
|
174
174
|
# (`lm.kwargs["api_key"]`). Never harvest it from the environment or the
|
|
175
175
|
# boto3 credential chain — auto-resolving and forwarding env credentials
|
|
176
176
|
# (especially AWS access key / secret / session token for a `bedrock/...`
|
|
@@ -211,7 +211,7 @@ def deploy(module, metric=None, timeout=300, metadata=None):
|
|
|
211
211
|
# Surface non-fatal deploy-time warnings (e.g. no provider credential
|
|
212
212
|
# resolves yet, so execute will 422 until one is configured). The
|
|
213
213
|
# server decides — it knows whether a provider exists — so the SDK only
|
|
214
|
-
# relays. Backward-compatible: absent on older servers
|
|
214
|
+
# relays. Backward-compatible: absent on older servers.
|
|
215
215
|
for warning in data.get("warnings") or []:
|
|
216
216
|
logger.warning("deploy: %s", warning)
|
|
217
217
|
|
|
@@ -51,7 +51,7 @@ class EvalHandler:
|
|
|
51
51
|
|
|
52
52
|
dspy, _ = callback._get_dspy()
|
|
53
53
|
|
|
54
|
-
# Create the eval run.
|
|
54
|
+
# Create the eval run. No id minted here — the server assigns
|
|
55
55
|
# it via the synchronous export_eval_start below, once the header
|
|
56
56
|
# fields (program/metric/dataset) are populated.
|
|
57
57
|
eval_run = models.EvalRunBuilder(
|
|
@@ -180,7 +180,7 @@ class EvalHandler:
|
|
|
180
180
|
f"dataset_size={eval_run.dataset_size}"
|
|
181
181
|
)
|
|
182
182
|
|
|
183
|
-
#
|
|
183
|
+
# The SDK doesn't mint the eval_run_id — the component behind
|
|
184
184
|
# the exporter (serve/observe, or deploy in-Lambda) assigns it. Obtain
|
|
185
185
|
# it synchronously now, before any eval trace ships, since every
|
|
186
186
|
# in-flight trace and the dict keys / span below are stamped with it.
|
|
@@ -11,6 +11,12 @@ from typing import TYPE_CHECKING
|
|
|
11
11
|
|
|
12
12
|
import httpx
|
|
13
13
|
|
|
14
|
+
# Bound by name, not reached as `httpx.HTTPStatusError`: the SDK's own tests patch this
|
|
15
|
+
# module's whole `httpx` attribute, so constructing the exception off that attribute
|
|
16
|
+
# raises "exceptions must derive from BaseException" under test while working in
|
|
17
|
+
# production. A direct import is the same class both ways.
|
|
18
|
+
from httpx import HTTPStatusError
|
|
19
|
+
|
|
14
20
|
from . import configuration
|
|
15
21
|
|
|
16
22
|
if TYPE_CHECKING:
|
|
@@ -118,6 +124,26 @@ def _validate_str_map(name, value):
|
|
|
118
124
|
return value
|
|
119
125
|
|
|
120
126
|
|
|
127
|
+
def _upstream_detail(resp: httpx.Response) -> str:
|
|
128
|
+
"""The server's own message for a failed execute, for the exception text.
|
|
129
|
+
|
|
130
|
+
Both shapes the platform emits: `{"error": {"message": …}}` (a data plane's refusal,
|
|
131
|
+
passed through by serve) and `{"error": "…"}` (serve's own writeError). Falls back to
|
|
132
|
+
the raw text, trimmed, because an unparseable body still beats no body — and returns
|
|
133
|
+
"" rather than guessing when there is nothing, so the caller omits the suffix.
|
|
134
|
+
"""
|
|
135
|
+
try:
|
|
136
|
+
data = resp.json()
|
|
137
|
+
except ValueError:
|
|
138
|
+
return (resp.text or "").strip()[:2000]
|
|
139
|
+
error = data.get("error") if isinstance(data, dict) else None
|
|
140
|
+
if isinstance(error, dict):
|
|
141
|
+
return str(error.get("message") or error).strip()[:2000]
|
|
142
|
+
if error:
|
|
143
|
+
return str(error).strip()[:2000]
|
|
144
|
+
return (resp.text or "").strip()[:2000]
|
|
145
|
+
|
|
146
|
+
|
|
121
147
|
def execute(deployment_id, timeout=30, metadata=None, tags=None, **inputs):
|
|
122
148
|
"""Execute a deployed DSPy program.
|
|
123
149
|
|
|
@@ -173,5 +199,20 @@ def execute(deployment_id, timeout=30, metadata=None, tags=None, **inputs):
|
|
|
173
199
|
data = resp.json()
|
|
174
200
|
raise RuntimeError(data.get("error", "Execution failed"))
|
|
175
201
|
|
|
176
|
-
resp.
|
|
202
|
+
if resp.status_code >= 400:
|
|
203
|
+
# Same exception `raise_for_status()` would raise — the type is a contract:
|
|
204
|
+
# this function's docstring promises it, and `tests/stress/_reliability.py`
|
|
205
|
+
# keys its upstream-outage tolerance on it, so turning a 5xx into anything
|
|
206
|
+
# else would make every provider hiccup fail a PR instead of skipping.
|
|
207
|
+
# Only the MESSAGE changes: raise_for_status reports the status and the URL
|
|
208
|
+
# and drops the body, and the body is where the data plane says what
|
|
209
|
+
# happened — serve passes it through verbatim (`WriteUpstreamError`). A
|
|
210
|
+
# sandbox refusal naming the missing module and the user's own file and line
|
|
211
|
+
# was arriving as a bare "Server error '500 Internal Server Error'".
|
|
212
|
+
detail = _upstream_detail(resp)
|
|
213
|
+
raise HTTPStatusError(
|
|
214
|
+
f"Execution failed with {resp.status_code} for {resp.request.url}" + (f": {detail}" if detail else ""),
|
|
215
|
+
request=resp.request,
|
|
216
|
+
response=resp,
|
|
217
|
+
)
|
|
177
218
|
return Prediction(**resp.json())
|
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
"""Which exporter the SDK is currently publishing through, and its lifecycle.
|
|
2
|
+
|
|
3
|
+
The HTTP implementation lives in :mod:`cmpnd.exporter_http`; this module holds
|
|
4
|
+
only the global that names the current one. The split is an import-weight
|
|
5
|
+
change: :mod:`cmpnd.callback` reaches this module for
|
|
6
|
+
:func:`ensure_exporter_started` and :func:`get_exporter` and nothing else, so
|
|
7
|
+
with the transport moved out the whole trace-production path — ``callback``,
|
|
8
|
+
``models``, ``context``, ``helpers``, ``configuration``, ``eval_handler`` —
|
|
9
|
+
imports no third-party module at all, and *defining* a trace no longer costs
|
|
10
|
+
httpx's import. The transport is imported on first export instead, by the
|
|
11
|
+
process that actually exports.
|
|
12
|
+
|
|
13
|
+
The global is duck-typed, and deliberately so: three callers replace it with an
|
|
14
|
+
in-memory collector to read traces back with no server in the loop —
|
|
15
|
+
``deploy/deploy/executor_handler.py``, ``deploy/deploy/iterate_handler.py`` and
|
|
16
|
+
``worker-sandbox/kernel/worker_kernel/run.py``. They assign
|
|
17
|
+
``cmpnd.exporter._exporter`` under ``_exporter_lock`` directly, which is why the
|
|
18
|
+
registry stays in this module while the implementation moves: an implementation
|
|
19
|
+
detail can be relocated, a seam three components already write to cannot.
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
from __future__ import annotations
|
|
23
|
+
|
|
24
|
+
import threading
|
|
25
|
+
from typing import TYPE_CHECKING, Any
|
|
26
|
+
|
|
27
|
+
from . import configuration
|
|
28
|
+
|
|
29
|
+
if TYPE_CHECKING:
|
|
30
|
+
# Redundant aliases, which is PEP 484's way of saying "re-exported" rather than
|
|
31
|
+
# "imported for local use". Without them a type checker is right to warn that
|
|
32
|
+
# `from cmpnd.exporter import BatchExporter` reaches a private import — and these
|
|
33
|
+
# two names are deliberately still part of this module's surface, so saying so
|
|
34
|
+
# here is the fix rather than silencing the warning at each callsite. The runtime
|
|
35
|
+
# half is `__getattr__` at the bottom of this file.
|
|
36
|
+
from .exporter_http import BatchExporter as BatchExporter
|
|
37
|
+
from .exporter_http import ExportItem as ExportItem
|
|
38
|
+
|
|
39
|
+
_exporter: BatchExporter | None = None
|
|
40
|
+
_exporter_lock = threading.Lock()
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def get_exporter() -> BatchExporter | None:
|
|
44
|
+
"""Get the global exporter instance."""
|
|
45
|
+
return _exporter
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def ensure_exporter_started() -> BatchExporter | None:
|
|
49
|
+
"""Ensure the exporter is started, creating it if necessary."""
|
|
50
|
+
global _exporter
|
|
51
|
+
|
|
52
|
+
with _exporter_lock:
|
|
53
|
+
if _exporter is None:
|
|
54
|
+
config = configuration.get_config()
|
|
55
|
+
if config:
|
|
56
|
+
# The one place the transport is imported at run time. An
|
|
57
|
+
# unconfigured SDK — and every collector-backed host, which
|
|
58
|
+
# installs its own exporter before the callback asks for one —
|
|
59
|
+
# never reaches it.
|
|
60
|
+
from .exporter_http import BatchExporter
|
|
61
|
+
|
|
62
|
+
_exporter = BatchExporter(config)
|
|
63
|
+
_exporter.start()
|
|
64
|
+
return _exporter
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def flush_exporter(timeout: float = 5.0) -> bool:
|
|
68
|
+
"""Force-publish everything the global exporter has queued, keeping it running.
|
|
69
|
+
|
|
70
|
+
A non-destructive alternative to :func:`shutdown_exporter` for a mid-run
|
|
71
|
+
reader that wants the just-produced trace to exist server-side before
|
|
72
|
+
fetching it — without tearing the exporter down (which would silently stop
|
|
73
|
+
later spans from publishing).
|
|
74
|
+
|
|
75
|
+
Returns ``True`` if the drain completed within ``timeout``, ``False`` if it
|
|
76
|
+
timed out or there is no exporter running — so a caller can avoid fetching a
|
|
77
|
+
trace that may not have been published yet, rather than assuming success.
|
|
78
|
+
"""
|
|
79
|
+
exp = get_exporter()
|
|
80
|
+
if exp:
|
|
81
|
+
return exp.flush(timeout)
|
|
82
|
+
return False
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def shutdown_exporter() -> None:
|
|
86
|
+
"""Shutdown the global exporter."""
|
|
87
|
+
global _exporter
|
|
88
|
+
|
|
89
|
+
with _exporter_lock:
|
|
90
|
+
if _exporter:
|
|
91
|
+
_exporter.shutdown()
|
|
92
|
+
_exporter = None
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
# The transport's two names, still importable from here (PEP 562). Moving the
|
|
96
|
+
# implementation was supposed to change what `import cmpnd.exporter` costs, not
|
|
97
|
+
# what it offers, and `from cmpnd.exporter import BatchExporter` is what callers
|
|
98
|
+
# already write. Resolving on attribute access is what keeps the two facts
|
|
99
|
+
# compatible — the import happens only if someone asks for the name.
|
|
100
|
+
_MOVED_TO_TRANSPORT = ("BatchExporter", "ExportItem")
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def __getattr__(name: str) -> Any:
|
|
104
|
+
if name in _MOVED_TO_TRANSPORT:
|
|
105
|
+
from . import exporter_http
|
|
106
|
+
|
|
107
|
+
return getattr(exporter_http, name)
|
|
108
|
+
raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
|