cmpnd 0.6.1__tar.gz → 0.6.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {cmpnd-0.6.1 → cmpnd-0.6.2}/CLAUDE.md +17 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/PKG-INFO +1 -1
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/__init__.py +2 -2
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/_gepa_patch.py +3 -3
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/_program_patch.py +1 -1
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/_rlm_patch.py +1 -1
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/callback.py +9 -9
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/cli/__init__.py +2 -2
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/cli/commands/admin.py +1 -1
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/cli/commands/org.py +3 -3
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/cli/commands/top.py +3 -3
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/cli/shell.py +2 -2
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/cli/trace_render.py +2 -2
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/configuration.py +2 -2
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/context.py +1 -1
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/deployment.py +5 -5
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/eval_handler.py +2 -2
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/exporter_http.py +13 -13
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/helpers.py +1 -1
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/identity.py +4 -4
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/models.py +11 -11
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/optimization.py +2 -2
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/optimizers/gepa.py +1 -1
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/optimizers/gepa_callback.py +7 -7
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/optimizers/resume.py +1 -1
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/optimizers/tracker.py +10 -10
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/packaging.py +11 -11
- {cmpnd-0.6.1 → cmpnd-0.6.2}/pyproject.toml +1 -1
- {cmpnd-0.6.1 → cmpnd-0.6.2}/uv.lock +54 -54
- {cmpnd-0.6.1 → cmpnd-0.6.2}/.gitignore +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/README.md +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/cli/api_client.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/cli/commands/__init__.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/cli/commands/auth.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/cli/commands/datasets.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/cli/commands/deployments.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/cli/commands/evals.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/cli/commands/login.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/cli/commands/optimizations.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/cli/commands/programs.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/cli/commands/traces.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/cli/credentials.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/cli/verbs.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/dataset_sync.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/datasets.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/decorators.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/execution.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/exporter.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/ir/__init__.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/ir/decode.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/ir/encode.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/ir/hash.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/ir/program.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/ir/reflection.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/ir/types.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/ir/xxh64.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/cmpnd/optimizers/__init__.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/docs/2026-04-02-code-review.md +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/docs/plans/2026-04-05-train-val-dataset-capture-design.md +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/docs/plans/2026-04-05-train-val-dataset-capture.md +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/docs/sdk-instrumentation.md +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/examples/local_ollama.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/scripts/capture_review_run.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/scripts/fixtures/review_run.json.gz +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/scripts/gepa_resume_bug.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/scripts/seed_review_run.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/stubs/dspy/__init__.pyi +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/stubs/dspy/primitives/__init__.pyi +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/stubs/dspy/primitives/prediction.pyi +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/stubs/dspy/utils/__init__.pyi +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/stubs/dspy/utils/callback.pyi +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/__init__.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/_exporter_helpers.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/_identity_helpers.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/e2e/__init__.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/e2e/committee_task.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/e2e/conftest.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/e2e/data/committee_train.ndjson.gz +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/e2e/data/committee_val.ndjson.gz +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/e2e/fake_lm.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/e2e/helpers.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/e2e/test_capture_prompts.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/e2e/test_composite_module.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/e2e/test_dataset_auto_creation.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/e2e/test_deploy_execute.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/e2e/test_deploy_programs.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/e2e/test_edge_cases.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/e2e/test_example_hash.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/e2e/test_gepa_optimization.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/e2e/test_lm_accuracy.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/e2e/test_module_tracing.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/e2e/test_optimization_lineage.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/e2e/test_optimization_metric_identity.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/e2e/test_optimize_failures.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/e2e/test_optimize_loop.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/e2e/test_optimize_review_committee.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/e2e/test_optimize_server_owned_completion.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/e2e/test_parallel_export.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/e2e/test_project_attribution.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/e2e/test_react_tool_spans.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/e2e/test_save_load.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/e2e/test_trace_structure.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/e2e/test_user_attribution.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/identity_vectors/01_predict_qa.json +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/identity_vectors/02_predict_renamed_field.json +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/identity_vectors/03_chain_of_thought_qa.json +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/identity_vectors/04_predict_richer_types.json +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/identity_vectors/05_composite_two_predicts.json +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/identity_vectors/06_multichaincomparison.json +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/identity_vectors/07_program_of_thought.json +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/identity_vectors/08_opaque_module.json +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/identity_vectors/09_predict_pydantic_field.json +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/identity_vectors/10_react.json +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/identity_vectors/11_refine.json +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/identity_vectors/12_retrieve.json +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/identity_vectors/13_knn.json +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/identity_vectors/14_parallel.json +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/identity_vectors/README.md +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/integration/__init__.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/integration/conftest.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/integration/test_cli_deploy_smoketest.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/integration/test_logs_renders_traces.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/integration/test_optimizing_progress.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/integration/test_unhappy_path.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/parity/__init__.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/parity/conftest.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/parity/seed_parity.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/parity/test_auth.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/parity/test_datasets.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/parity/test_deployments.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/parity/test_evals.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/parity/test_health.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/parity/test_modules.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/parity/test_optimizations.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/parity/test_programs.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/parity/test_signatures.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/parity/test_trace_streaming.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/parity/test_traces.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/stress/__init__.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/stress/_provider.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/stress/_reliability.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/stress/mint_bedrock_token.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/stress/test_deploy_programs.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/stress/test_multi_client_stress.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/stress/test_optimize_client_committee.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/stress/test_optimize_loop.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/stress/test_optimize_loop_wasm.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/stress/test_optimize_stress.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/stress/test_real_provider_twins.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/stress/test_reliability.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/stress/test_reliability_classify.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_callback.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_cli/__init__.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_cli/conftest.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_cli/test_admin.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_cli/test_api_client.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_cli/test_auth.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_cli/test_credentials.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_cli/test_datasets.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_cli/test_deployments.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_cli/test_evals.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_cli/test_login.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_cli/test_main.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_cli/test_optimizations.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_cli/test_org.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_cli/test_parity.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_cli/test_programs.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_cli/test_shell.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_cli/test_top.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_cli/test_traces.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_clustering/__init__.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_clustering/conftest.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_clustering/test_edge_cases.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_clustering/test_hash_consistency.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_clustering/test_type_conversion.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_cmp551_grouping_identity.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_configuration.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_context.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_context_propagation.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_dataset_polling.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_dataset_sync.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_datasets.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_decorators.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_deploy.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_eval.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_eval_start.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_exception_paths.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_execute.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_exporter.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_exporter_real.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_fake_lm.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_gepa_callback.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_gepa_integration.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_gepa_resume.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_identity.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_import_weight.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_integration.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_ir_encoder_properties.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_ir_golden_vectors.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_ir_roundtrip.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_models.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_optimization_progress.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_optimization_start.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_optimize.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_optimize_lift.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_optimize_terminal_failure.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_packaging.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_packaging_callable_source.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_packaging_program_source.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_program_lineage_persist.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_project_payload.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_reflector_properties.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_reflector_smoke.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_repro_reactv2_bugs.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_rlm_patch.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_schemas.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_seed_review_run.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_stats_export.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_trace_render.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_tracked_gepa.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_tracker_enrich.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_truncation.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/tests/test_xxh64.py +0 -0
- {cmpnd-0.6.1 → cmpnd-0.6.2}/todo.md +0 -0
|
@@ -2,6 +2,23 @@
|
|
|
2
2
|
|
|
3
3
|
Part of the CMPND monorepo. See `../CLAUDE.md` for cross-cutting rules (hash architecture, null handling, testing philosophy).
|
|
4
4
|
|
|
5
|
+
## This code is released publicly — no private references
|
|
6
|
+
|
|
7
|
+
Unlike the rest of the monorepo (private), the `cmpnd` package ships to PyPI:
|
|
8
|
+
`python-sdk/cmpnd/**` is **public source**. Anything written there is
|
|
9
|
+
world-readable in the published wheel. So **never put private information in the
|
|
10
|
+
shipped package** — no internal issue-tracker IDs (`CMP-…`), internal URLs or
|
|
11
|
+
dashboards, customer/org names, infra hostnames, or secrets — and that includes
|
|
12
|
+
comments, docstrings, `--help`/CLI strings, error messages, and log lines, not
|
|
13
|
+
just code. Reference an internal ticket in the **commit message or PR
|
|
14
|
+
description** (those live in the private repo), never in the source that ships.
|
|
15
|
+
|
|
16
|
+
This applies only to the shipped package (`python-sdk/cmpnd/`). Tests
|
|
17
|
+
(`python-sdk/tests/`), `docs/`, and `docs/plans/` are not published in the
|
|
18
|
+
wheel, but prefer keeping issue IDs out of docstrings you might later move into
|
|
19
|
+
the package. When you cite prior work in shipped code, describe the behavior,
|
|
20
|
+
not the ticket.
|
|
21
|
+
|
|
5
22
|
## Documentation discipline (docs are as-built, updated in the same commit)
|
|
6
23
|
|
|
7
24
|
This mirrors the practice that has kept `serve/docs/` in sync with the Go
|
|
@@ -131,13 +131,13 @@ def auto_instrument() -> CmpndCallback:
|
|
|
131
131
|
|
|
132
132
|
# Persist the compiled-from lineage tag across state-only save/load so a
|
|
133
133
|
# continued optimization still stamps compiled_from_run_id after a disk
|
|
134
|
-
# reload / process restart
|
|
134
|
+
# reload / process restart.
|
|
135
135
|
from . import _program_patch
|
|
136
136
|
|
|
137
137
|
_program_patch.patch_program_state()
|
|
138
138
|
|
|
139
139
|
# Emit tool spans for dspy.RLM tool calls, which otherwise bypass
|
|
140
|
-
# dspy.Tool.__call__'s callbacks and produce no spans
|
|
140
|
+
# dspy.Tool.__call__'s callbacks and produce no spans.
|
|
141
141
|
from . import _rlm_patch
|
|
142
142
|
|
|
143
143
|
_rlm_patch.patch_rlm_tools()
|
|
@@ -84,7 +84,7 @@ def patch_gepa() -> None:
|
|
|
84
84
|
|
|
85
85
|
# Build the config dict from the GEPA instance, then hand the live
|
|
86
86
|
# program to the shared enrichment helper (one definition of the
|
|
87
|
-
# optimization_run payload shape, reused by the deploy path
|
|
87
|
+
# optimization_run payload shape, reused by the deploy path).
|
|
88
88
|
config = {}
|
|
89
89
|
for key in ["auto", "max_metric_calls", "max_full_evals", "seed", "log_dir"]:
|
|
90
90
|
val = getattr(self, key, None)
|
|
@@ -97,7 +97,7 @@ def patch_gepa() -> None:
|
|
|
97
97
|
except Exception:
|
|
98
98
|
pass
|
|
99
99
|
|
|
100
|
-
# Restore the up-front budget estimate
|
|
100
|
+
# Restore the up-front budget estimate: for an `auto`
|
|
101
101
|
# preset `self.max_metric_calls` is None until GEPA's first budget
|
|
102
102
|
# event, so the initial "running" record shipped 0 and the progress
|
|
103
103
|
# bar / ETA had no denominator. Estimate now instead.
|
|
@@ -160,7 +160,7 @@ def patch_gepa() -> None:
|
|
|
160
160
|
except Exception as e:
|
|
161
161
|
if callback is not None: # pyright: ignore[reportUnnecessaryComparison]
|
|
162
162
|
# Flush a terminal 'failed', not just in-memory state — else the
|
|
163
|
-
# running row start() sent lingers forever
|
|
163
|
+
# running row start() sent lingers forever.
|
|
164
164
|
callback._tracker.finalize_error(e)
|
|
165
165
|
raise
|
|
166
166
|
|
|
@@ -17,7 +17,7 @@ the state-only path, which is the one DSPy steers users toward.)
|
|
|
17
17
|
|
|
18
18
|
DSPy exposes no extension point for out-of-band instance state (no
|
|
19
19
|
``get_extra_state`` / attribute allowlist; ``metadata`` is DSPy-owned). Rather
|
|
20
|
-
than change DSPy upstream
|
|
20
|
+
than change DSPy upstream, ``patch_program_state()`` wraps
|
|
21
21
|
``dump_state`` / ``load_state`` on the two classes that define them — the
|
|
22
22
|
composite ``Module`` path and the bare ``Predict`` path — to round-trip the tag
|
|
23
23
|
under a namespaced carrier key. Installed lazily from
|
|
@@ -71,7 +71,7 @@ class _CallbackBase:
|
|
|
71
71
|
self._capture_inputs = (
|
|
72
72
|
capture_inputs if capture_inputs is not None else (config.capture_inputs if config else True)
|
|
73
73
|
)
|
|
74
|
-
# Emit an in_progress start for every root trace, not just RLM
|
|
74
|
+
# Emit an in_progress start for every root trace, not just RLM.
|
|
75
75
|
self._stream_trace_starts = (
|
|
76
76
|
stream_trace_starts if stream_trace_starts is not None else (config.stream_trace_starts if config else True)
|
|
77
77
|
)
|
|
@@ -117,7 +117,7 @@ class _CallbackBase:
|
|
|
117
117
|
self._eval_handler = eval_handler.EvalHandler(self)
|
|
118
118
|
|
|
119
119
|
# Active optimization run ID - stored on instance for same thread-safety reason.
|
|
120
|
-
# Prefixed text id (`op_…`), minted via cmpnd.identity
|
|
120
|
+
# Prefixed text id (`op_…`), minted via cmpnd.identity.
|
|
121
121
|
self._active_optimization_run_id: str | None = None
|
|
122
122
|
|
|
123
123
|
# Token usage from LM calls without a trace context (e.g., GEPA reflection calls)
|
|
@@ -188,7 +188,7 @@ class _ModuleMixin(_CallbackBase):
|
|
|
188
188
|
context.set_current_trace(trace)
|
|
189
189
|
|
|
190
190
|
# Carry the root module's type so serve can classify the trace at
|
|
191
|
-
# start
|
|
191
|
+
# start — before any span lands — instead of inferring it
|
|
192
192
|
# from span shape later.
|
|
193
193
|
trace.root_span_type = span_type.value
|
|
194
194
|
# RLM traces use streaming export (per-span); everything else
|
|
@@ -197,7 +197,7 @@ class _ModuleMixin(_CallbackBase):
|
|
|
197
197
|
trace.streaming = True
|
|
198
198
|
self._streaming_traces[str(trace.trace_id)] = trace.streaming
|
|
199
199
|
|
|
200
|
-
# Emit an in_progress start row for EVERY root trace
|
|
200
|
+
# Emit an in_progress start row for EVERY root trace, not
|
|
201
201
|
# just streaming/RLM ones, so a non-RLM program is visible the moment
|
|
202
202
|
# it starts (`traces search --status in_progress`, a lookupable id)
|
|
203
203
|
# instead of only appearing at finalize. `trace.streaming` still
|
|
@@ -251,7 +251,7 @@ class _ModuleMixin(_CallbackBase):
|
|
|
251
251
|
# Surface the previous turn's code output on THIS (turn N+1)
|
|
252
252
|
# span so a streaming reader can fill in turn N's notebook
|
|
253
253
|
# output mid-run, rather than waiting for the whole trajectory
|
|
254
|
-
# on the RLM parent span at finalize
|
|
254
|
+
# on the RLM parent span at finalize. The prior
|
|
255
255
|
# turn's span has already ended + been exported, so it rides
|
|
256
256
|
# here instead; the RLM's final turn (no successor) still
|
|
257
257
|
# resolves at finalize. Capped like an output field so a large
|
|
@@ -433,7 +433,7 @@ class _ModuleMixin(_CallbackBase):
|
|
|
433
433
|
trace._batch_spans = trace_spans
|
|
434
434
|
exp.export_trace(trace)
|
|
435
435
|
|
|
436
|
-
# Correlate the returned Prediction to its trace
|
|
436
|
+
# Correlate the returned Prediction to its trace.
|
|
437
437
|
# Stored under a leading underscore so it stays a plain instance
|
|
438
438
|
# attribute — a bare name would route into dspy.Prediction's field
|
|
439
439
|
# store and surface as a phantom output field. Read it back via the
|
|
@@ -785,7 +785,7 @@ class _LMMixin(_CallbackBase):
|
|
|
785
785
|
# new instructions, between evaluations) and metric (an LLM-as-judge /
|
|
786
786
|
# semantic metric scoring outputs, during evaluation). Rather than drop
|
|
787
787
|
# them, wrap each in its own trace linked to the optimization run so it
|
|
788
|
-
# shows up alongside the student traces
|
|
788
|
+
# shows up alongside the student traces. serve's
|
|
789
789
|
# TracedTokensForRun then accounts for their cost from the span, so these
|
|
790
790
|
# must NOT also be added to _orphaned_lm_usage / untraced_lm_usage — that
|
|
791
791
|
# would double-count them in the optimization total.
|
|
@@ -821,7 +821,7 @@ class _LMMixin(_CallbackBase):
|
|
|
821
821
|
is running — instead of assuming every untraced call is reflection (that
|
|
822
822
|
would mislabel judge calls). The trace is named with the umbrella "GEPA";
|
|
823
823
|
the specific model stays on the span's model_name, and `gepa.phase`
|
|
824
|
-
carries the distinction
|
|
824
|
+
carries the distinction. Each becomes a standalone one-span
|
|
825
825
|
trace so the prompt, response, tokens and cost are visible alongside the
|
|
826
826
|
student traces instead of being reduced to an aggregate token count.
|
|
827
827
|
"""
|
|
@@ -842,7 +842,7 @@ class _LMMixin(_CallbackBase):
|
|
|
842
842
|
# context stack, so on_lm_start stamped this span with a parent_span_id for
|
|
843
843
|
# a span that isn't in this standalone one-span trace. This span IS the
|
|
844
844
|
# trace's root, so force it parentless and pin it to this trace so tree/root
|
|
845
|
-
# reconstruction works
|
|
845
|
+
# reconstruction works.
|
|
846
846
|
span.parent_span_id = None
|
|
847
847
|
span.trace_id = trace.trace_id
|
|
848
848
|
trace.signature_name = span.model_name or phase
|
|
@@ -87,7 +87,7 @@ def main() -> None:
|
|
|
87
87
|
auth.register(subparsers)
|
|
88
88
|
login.register(subparsers)
|
|
89
89
|
|
|
90
|
-
# Org membership (owner-facing approval queue
|
|
90
|
+
# Org membership (owner-facing approval queue). Always
|
|
91
91
|
# registered — the server gates each action on org ownership, so a
|
|
92
92
|
# non-owner just gets a 403, not a hidden command.
|
|
93
93
|
org.register(subparsers)
|
|
@@ -99,7 +99,7 @@ def main() -> None:
|
|
|
99
99
|
if os.environ.get("CMPND_CLI_ADMIN") == "1":
|
|
100
100
|
admin.register(subparsers)
|
|
101
101
|
|
|
102
|
-
#
|
|
102
|
+
# Resource subcommands are dev-only until productized.
|
|
103
103
|
# Set CMPND_CLI_DEV=1 to expose programs, traces, optimizations,
|
|
104
104
|
# evals, datasets, and deployments in --help.
|
|
105
105
|
if os.environ.get("CMPND_CLI_DEV") == "1":
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
"""Platform-operator commands, hidden behind CMPND_CLI_ADMIN=1.
|
|
2
2
|
|
|
3
3
|
`cmpnd admin org create|set-owner|list|get|set-limit|set-github-login|elect-owner`
|
|
4
|
-
is the closed-beta org bootstrap and steward surface
|
|
4
|
+
is the closed-beta org bootstrap and steward surface:
|
|
5
5
|
stand up a shared org with its GitHub auto-join key, anoint its first owner,
|
|
6
6
|
adjust its member cap or GitHub link, and grant ownership from the election
|
|
7
7
|
queue. `cmpnd admin user …` is the operator's user-lifecycle surface (list,
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
"""Org membership + ownership commands (customer-facing).
|
|
2
2
|
|
|
3
|
-
`cmpnd org members …` is the owner's delegated membership surface
|
|
4
|
-
|
|
3
|
+
`cmpnd org members …` is the owner's delegated membership surface: list
|
|
4
|
+
the GitHub users who auto-joined and are awaiting approval and
|
|
5
5
|
admit/decline them (the pull direction), and issue/list/revoke single-use
|
|
6
6
|
invite links for teammates you want to add (the push direction — `invite`
|
|
7
7
|
prints a URL you share out-of-band; the invitee joins as a member when they
|
|
@@ -15,7 +15,7 @@ hit /api/v1/org/ownership-requests and are self-scoped to the calling key's
|
|
|
15
15
|
user — no owner check (a platform admin grants ownership from the election
|
|
16
16
|
queue; the request is just recorded intent).
|
|
17
17
|
|
|
18
|
-
`cmpnd org keys list|create|revoke` is your own API-key surface
|
|
18
|
+
`cmpnd org keys list|create|revoke` is your own API-key surface:
|
|
19
19
|
list the keys you've minted, mint a new scoped key (the response carries the
|
|
20
20
|
one-time token — the server stores only its hash), and revoke one. These hit
|
|
21
21
|
/api/v1/org/keys and are self-scoped to the calling key's user — you only ever
|
|
@@ -113,8 +113,8 @@ def _hash_fallback(sig: Any) -> str:
|
|
|
113
113
|
is what someone would copy/paste. Names should be the common case once the
|
|
114
114
|
Program registry has caught up — this is just the fallback.
|
|
115
115
|
"""
|
|
116
|
-
# program_signature_hash now arrives as a decimal string on the wire
|
|
117
|
-
#
|
|
116
|
+
# program_signature_hash now arrives as a decimal string on the wire;
|
|
117
|
+
# int(sig) below handles both forms. "0" is the unset sentinel
|
|
118
118
|
# alongside numeric 0 / None / "".
|
|
119
119
|
if sig in (None, "", 0, "0"):
|
|
120
120
|
return "—"
|
|
@@ -379,7 +379,7 @@ def _render_status(d: dict[str, Any], endpoint: str) -> None:
|
|
|
379
379
|
print(f" Deploy prog: {deploy_program_id}")
|
|
380
380
|
print(f" Status: {_status(d.get('status', '—'))} ({age})")
|
|
381
381
|
print(f" Created: {d.get('created_at', '—')}")
|
|
382
|
-
# The caller's deploy(metadata=…) blob, if any
|
|
382
|
+
# The caller's deploy(metadata=…) blob, if any — the
|
|
383
383
|
# checkpoint/cursor that produced this program. Omitted when empty ({}).
|
|
384
384
|
metadata = d.get("metadata") or {}
|
|
385
385
|
if metadata:
|
|
@@ -92,7 +92,7 @@ def _logged_out_banner(endpoint: str) -> str:
|
|
|
92
92
|
f"{_c('cmpnd shell', 'bold')} — not signed in ({endpoint}).\n"
|
|
93
93
|
f" Set {_c('CMPND_API_KEY', 'bold')} (and optionally "
|
|
94
94
|
f"{_c('CMPND_ENDPOINT', 'bold')}), then restart.\n"
|
|
95
|
-
f" Browser-based {_c('cmpnd login', 'bold')} is on the way
|
|
95
|
+
f" Browser-based {_c('cmpnd login', 'bold')} is on the way.\n"
|
|
96
96
|
)
|
|
97
97
|
|
|
98
98
|
|
|
@@ -300,7 +300,7 @@ def _parse_signature_inputs(program_name: str | None) -> list[str]:
|
|
|
300
300
|
``run(body=None, **kwargs)`` form for those.
|
|
301
301
|
|
|
302
302
|
Stopgap: the deployment record only carries ``program_name`` as a
|
|
303
|
-
string today, not structured input/output fields.
|
|
303
|
+
string today, not structured input/output fields. A follow-up tracks
|
|
304
304
|
surfacing those on the serializer (sourced from what the deploy
|
|
305
305
|
service already produces for ``/openapi``). Once that lands, delete
|
|
306
306
|
this function and key off the structured fields instead — they'll
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
"""Render a trace's span forest as a tree or an ASCII timeline.
|
|
2
2
|
|
|
3
|
-
Consumed by ``cmpnd traces show <id
|
|
3
|
+
Consumed by ``cmpnd traces show <id>``. The raw
|
|
4
4
|
``GET /api/v1/traces/{id}`` response — ``{"trace": {...}, "spans": [...]}``
|
|
5
5
|
with flat, raw-column-named spans — is turned into something readable when the
|
|
6
6
|
caller opts into ``--tree`` / ``--timeline`` (``show`` is JSON by default, like
|
|
@@ -241,7 +241,7 @@ def span_line(span: dict[str, Any], *, color: bool) -> str:
|
|
|
241
241
|
"""One span rendered as a single line: '<type label> <name> · <duration> ·
|
|
242
242
|
<tokens> · <model>', mirroring serve's span-badge + span-name, then the same
|
|
243
243
|
duration string. Used both as a tree node body and as `traces follow`'s
|
|
244
|
-
per-span line
|
|
244
|
+
per-span line."""
|
|
245
245
|
# Two visually distinct segments: the span-type label carries the status
|
|
246
246
|
# hue (green/red/yellow), the span name is bold default-foreground. Bold on
|
|
247
247
|
# the terminal's own fg reads on any background; the two never share a color.
|
|
@@ -78,12 +78,12 @@ class Config:
|
|
|
78
78
|
max_queue_size: int = 10000
|
|
79
79
|
max_export_workers: int = 4
|
|
80
80
|
|
|
81
|
-
# Emit an in_progress trace_start for every root trace
|
|
81
|
+
# Emit an in_progress trace_start for every root trace, not just
|
|
82
82
|
# RLM, so non-RLM programs (Predict/ReAct/CoT/optimizer students) are
|
|
83
83
|
# visible while running instead of appearing only at finalize. The exporter
|
|
84
84
|
# coalesces start + batch in one flush, so a fast trace still pays a single
|
|
85
85
|
# insert. Set False to opt out of the extra start (traces then reappear only
|
|
86
|
-
# at finalize, the
|
|
86
|
+
# at finalize, the prior behavior).
|
|
87
87
|
stream_trace_starts: bool = True
|
|
88
88
|
|
|
89
89
|
# Feature flags
|
|
@@ -20,7 +20,7 @@ _span_stack: ContextVar[list[models.SpanBuilder] | None] = ContextVar("cmpnd_spa
|
|
|
20
20
|
_current_eval_run: ContextVar[models.EvalRunBuilder | None] = ContextVar("cmpnd_current_eval_run", default=None)
|
|
21
21
|
|
|
22
22
|
# Current optimization run context (set during optimization, used to link
|
|
23
|
-
# evals). Prefixed text id (`op_…`), minted via cmpnd.identity
|
|
23
|
+
# evals). Prefixed text id (`op_…`), minted via cmpnd.identity.
|
|
24
24
|
_current_optimization_run_id: ContextVar[str | None] = ContextVar("cmpnd_current_optimization_run_id", default=None)
|
|
25
25
|
|
|
26
26
|
|
|
@@ -20,7 +20,7 @@ from . import configuration, execution, identity, packaging
|
|
|
20
20
|
|
|
21
21
|
logger = logging.getLogger(__name__)
|
|
22
22
|
|
|
23
|
-
#
|
|
23
|
+
# The deploy metadata blob is a free-form JSON object committed with
|
|
24
24
|
# the deployment record (the checkpoint/cursor that produced the program). Cap
|
|
25
25
|
# its size — a checkpoint/cursor is a pointer, not a payload. Mirrors serve's
|
|
26
26
|
# maxMetadataBytes so the call-site error matches the server's 413.
|
|
@@ -127,7 +127,7 @@ def deploy(module, metric=None, timeout=300, metadata=None):
|
|
|
127
127
|
metadata: Optional free-form structured dict committed atomically with
|
|
128
128
|
the deployment record — somewhere to put the checkpoint/cursor that
|
|
129
129
|
produced this program, so the deployment is self-describing for a
|
|
130
|
-
continual-learning loop
|
|
130
|
+
continual-learning loop. Must be a JSON-serializable dict
|
|
131
131
|
with string keys, within 64 KiB. Read it back via the deployment
|
|
132
132
|
record (``cmpnd.status`` / the deployment GET API).
|
|
133
133
|
|
|
@@ -155,7 +155,7 @@ def deploy(module, metric=None, timeout=300, metadata=None):
|
|
|
155
155
|
# slot and reject an oversized deploy before signing it (Slice E).
|
|
156
156
|
"declared_content_length": len(zip_bytes),
|
|
157
157
|
}
|
|
158
|
-
#
|
|
158
|
+
# Carry the caller's metadata blob (validated at the call site) as a
|
|
159
159
|
# reserved top-level body key, committed with the deployment record server-side.
|
|
160
160
|
if metadata is not None:
|
|
161
161
|
_validate_deploy_metadata(metadata)
|
|
@@ -170,7 +170,7 @@ def deploy(module, metric=None, timeout=300, metadata=None):
|
|
|
170
170
|
|
|
171
171
|
# Forward LM credentials so the server can auto-create a Provider+ProviderKey.
|
|
172
172
|
#
|
|
173
|
-
#
|
|
173
|
+
# Credential policy: forward ONLY a key explicitly set on the LM
|
|
174
174
|
# (`lm.kwargs["api_key"]`). Never harvest it from the environment or the
|
|
175
175
|
# boto3 credential chain — auto-resolving and forwarding env credentials
|
|
176
176
|
# (especially AWS access key / secret / session token for a `bedrock/...`
|
|
@@ -211,7 +211,7 @@ def deploy(module, metric=None, timeout=300, metadata=None):
|
|
|
211
211
|
# Surface non-fatal deploy-time warnings (e.g. no provider credential
|
|
212
212
|
# resolves yet, so execute will 422 until one is configured). The
|
|
213
213
|
# server decides — it knows whether a provider exists — so the SDK only
|
|
214
|
-
# relays. Backward-compatible: absent on older servers
|
|
214
|
+
# relays. Backward-compatible: absent on older servers.
|
|
215
215
|
for warning in data.get("warnings") or []:
|
|
216
216
|
logger.warning("deploy: %s", warning)
|
|
217
217
|
|
|
@@ -51,7 +51,7 @@ class EvalHandler:
|
|
|
51
51
|
|
|
52
52
|
dspy, _ = callback._get_dspy()
|
|
53
53
|
|
|
54
|
-
# Create the eval run.
|
|
54
|
+
# Create the eval run. No id minted here — the server assigns
|
|
55
55
|
# it via the synchronous export_eval_start below, once the header
|
|
56
56
|
# fields (program/metric/dataset) are populated.
|
|
57
57
|
eval_run = models.EvalRunBuilder(
|
|
@@ -180,7 +180,7 @@ class EvalHandler:
|
|
|
180
180
|
f"dataset_size={eval_run.dataset_size}"
|
|
181
181
|
)
|
|
182
182
|
|
|
183
|
-
#
|
|
183
|
+
# The SDK doesn't mint the eval_run_id — the component behind
|
|
184
184
|
# the exporter (serve/observe, or deploy in-Lambda) assigns it. Obtain
|
|
185
185
|
# it synchronously now, before any eval trace ships, since every
|
|
186
186
|
# in-flight trace and the dict keys / span below are stamped with it.
|
|
@@ -30,12 +30,12 @@ if TYPE_CHECKING:
|
|
|
30
30
|
from .optimizers import tracker
|
|
31
31
|
|
|
32
32
|
# Named for the module this code used to live in, not for the one it lives in
|
|
33
|
-
# now: the per-flush "cmpnd: published N records" line
|
|
33
|
+
# now: the per-flush "cmpnd: published N records" line is a
|
|
34
34
|
# user-visible INFO log, and a logger rename would silently break anyone who
|
|
35
35
|
# filters or levels on `cmpnd.exporter`.
|
|
36
36
|
logger = logging.getLogger("cmpnd.exporter")
|
|
37
37
|
|
|
38
|
-
# Every exporter body ships gzipped
|
|
38
|
+
# Every exporter body ships gzipped. The motivation is bandwidth on
|
|
39
39
|
# multithreaded GEPA runs over cached LM calls — the same pressure that used to
|
|
40
40
|
# drive thinning adapter spans, which stripped the inputs/outputs people want
|
|
41
41
|
# when debugging an adapter. Compressing the wire instead lets the SDK send the
|
|
@@ -88,7 +88,7 @@ class BatchExporter:
|
|
|
88
88
|
self._transport = transport
|
|
89
89
|
self._queue: queue.Queue[ExportItem] = queue.Queue(maxsize=config.max_queue_size)
|
|
90
90
|
self._shutdown = threading.Event()
|
|
91
|
-
# Non-destructive flush
|
|
91
|
+
# Non-destructive flush: a mid-run reader signals the export
|
|
92
92
|
# loop to drain its current batch without ending the loop, then waits
|
|
93
93
|
# for confirmation. Distinct from _shutdown, which ends the loop.
|
|
94
94
|
#
|
|
@@ -104,7 +104,7 @@ class BatchExporter:
|
|
|
104
104
|
self._client: httpx.Client | None = None
|
|
105
105
|
self._executor: ThreadPoolExecutor | None = None
|
|
106
106
|
|
|
107
|
-
# Synchronous client for the blocking "start" round-trips
|
|
107
|
+
# Synchronous client for the blocking "start" round-trips.
|
|
108
108
|
# Distinct from the background _client (which lives on the export
|
|
109
109
|
# thread); created lazily on first start so unconfigured exporters
|
|
110
110
|
# never open a socket. httpx.Client is safe to share across threads.
|
|
@@ -231,7 +231,7 @@ class BatchExporter:
|
|
|
231
231
|
|
|
232
232
|
def export_eval_start(self, eval_run: models.EvalRunBuilder) -> str:
|
|
233
233
|
"""Create the "running" eval record server-side and return the
|
|
234
|
-
server-assigned ``eval_run_id
|
|
234
|
+
server-assigned ``eval_run_id``.
|
|
235
235
|
|
|
236
236
|
Synchronous and blocking, mirroring ``export_optimization_start``: the
|
|
237
237
|
id must exist before the first eval trace ships. The SDK sends the
|
|
@@ -244,7 +244,7 @@ class BatchExporter:
|
|
|
244
244
|
def export_eval_run(self, eval_run: models.EvalRunBuilder) -> None:
|
|
245
245
|
"""Queue an eval-run completion for export (non-blocking).
|
|
246
246
|
|
|
247
|
-
|
|
247
|
+
This is the completion half of the start/complete split — it
|
|
248
248
|
PATCHes the running record (created by export_eval_start) with results.
|
|
249
249
|
Async is fine; no id is needed back.
|
|
250
250
|
"""
|
|
@@ -268,7 +268,7 @@ class BatchExporter:
|
|
|
268
268
|
|
|
269
269
|
def export_optimization_start(self, optimization_run: tracker.OptimizationTracker) -> str:
|
|
270
270
|
"""Create the "running" optimization record server-side and return the
|
|
271
|
-
server-assigned ``optimization_run_id
|
|
271
|
+
server-assigned ``optimization_run_id``.
|
|
272
272
|
|
|
273
273
|
Synchronous and blocking: the id must exist before the first trace
|
|
274
274
|
ships, since every in-flight trace is stamped with it. The SDK no
|
|
@@ -378,7 +378,7 @@ class BatchExporter:
|
|
|
378
378
|
batch = []
|
|
379
379
|
last_flush = time.time()
|
|
380
380
|
|
|
381
|
-
# Non-destructive flush request
|
|
381
|
+
# Non-destructive flush request: drain everything
|
|
382
382
|
# currently pending — the local batch (items already pulled off
|
|
383
383
|
# the queue, which a bare get_nowait() drain would miss) plus
|
|
384
384
|
# any items still sitting in the queue — then confirm. The loop
|
|
@@ -446,7 +446,7 @@ class BatchExporter:
|
|
|
446
446
|
return
|
|
447
447
|
|
|
448
448
|
# Snapshot the counters so the end-of-flush confirmation reports what the
|
|
449
|
-
# server actually accepted, not what we attempted
|
|
449
|
+
# server actually accepted, not what we attempted. Safe to read
|
|
450
450
|
# without the lock here: _flush_batch runs only on the export thread (or
|
|
451
451
|
# the main thread during shutdown, after the export thread is joined), and
|
|
452
452
|
# all per-flush increments happen on dispatch threads we join via
|
|
@@ -506,7 +506,7 @@ class BatchExporter:
|
|
|
506
506
|
new_spans[trace_id] = []
|
|
507
507
|
new_spans[trace_id].append(span_dict)
|
|
508
508
|
|
|
509
|
-
# Coalesce start + batch within one flush
|
|
509
|
+
# Coalesce start + batch within one flush: if a trace's full
|
|
510
510
|
# batch trace is in this same flush, sending it upserts the row to its
|
|
511
511
|
# final status directly (serve traceUpsertSQL, ON CONFLICT on
|
|
512
512
|
# org_id,trace_id,start_time), so the in_progress start POST is
|
|
@@ -574,8 +574,8 @@ class BatchExporter:
|
|
|
574
574
|
with self._stats_lock:
|
|
575
575
|
self._error_count += 1
|
|
576
576
|
|
|
577
|
-
# One default-verbosity confirmation per flush that did something
|
|
578
|
-
#
|
|
577
|
+
# One default-verbosity confirmation per flush that did something,
|
|
578
|
+
# so a user at INFO can tell "configured" from "actually
|
|
579
579
|
# published". Reports the delta in accepted records (a 2xx increments
|
|
580
580
|
# _exported_count) — NOT what we attempted — so a rejected or dropped
|
|
581
581
|
# item never reads as a false success; failures show in the dropped/
|
|
@@ -679,7 +679,7 @@ class BatchExporter:
|
|
|
679
679
|
self._error_count += len(stats_to_send)
|
|
680
680
|
|
|
681
681
|
def _send_eval_runs(self, eval_runs_to_send: list[dict[str, Any]]) -> None:
|
|
682
|
-
#
|
|
682
|
+
# Completion is a PATCH to the running record created by the
|
|
683
683
|
# synchronous export_eval_start, keyed by the server-assigned id.
|
|
684
684
|
client = self._require_client()
|
|
685
685
|
for eval_run_dict in eval_runs_to_send:
|
|
@@ -13,7 +13,7 @@ def truncate_field(value: Any, max_size: int, depth: int = 0) -> Any:
|
|
|
13
13
|
|
|
14
14
|
Thin delegate to ``models.truncate_field`` — the one truncator, homed in
|
|
15
15
|
``models`` so ``SpanBuilder.to_dict`` can reach it without the
|
|
16
|
-
``helpers``→``models`` circular import
|
|
16
|
+
``helpers``→``models`` circular import. The callback capture path
|
|
17
17
|
(``clean_inputs``/``clean_outputs``/``truncate_outputs``) keeps calling this.
|
|
18
18
|
``depth`` forwards so a caller entering per-field-value truncation can mark
|
|
19
19
|
the value as nested (``depth=1``) and get the dict key-cap, matching
|
|
@@ -19,7 +19,7 @@ import types as _stdtypes
|
|
|
19
19
|
import typing
|
|
20
20
|
from typing import Any
|
|
21
21
|
|
|
22
|
-
#
|
|
22
|
+
# The SDK no longer mints eval/optimization/example ids. They are
|
|
23
23
|
# server-assigned and opaque to the SDK — the server (observe UUIDs, serve
|
|
24
24
|
# `op_`/`ev_`/`ex_` base62) mints on the synchronous start POST / on ingest,
|
|
25
25
|
# and the SDK adopts whatever it returns. The old new_eval_run_id /
|
|
@@ -105,7 +105,7 @@ def format_signature_hash(value: int | None) -> str | None:
|
|
|
105
105
|
unsafe: any consumer that decodes through an untyped path floats it and
|
|
106
106
|
loses precision above 2^53 (Go ``any``/``map[string]any``, JavaScript
|
|
107
107
|
natively). Carrying it as a string makes the safe representation automatic
|
|
108
|
-
end to end
|
|
108
|
+
end to end. Storage stays signed int64; only the wire value is a
|
|
109
109
|
string. ``None`` stays ``None`` (nullable optimization hash); the trace/span
|
|
110
110
|
``0`` sentinel renders as ``"0"``.
|
|
111
111
|
"""
|
|
@@ -211,7 +211,7 @@ def compute_valset_identity_hash(example_hashes: list[str]) -> str:
|
|
|
211
211
|
(``_example_input_hash`` emits ``""``); such a set has no trustworthy identity,
|
|
212
212
|
so return "" (excluded from grouping) rather than hashing the blanks —
|
|
213
213
|
otherwise two unrelated all-blank sets of the same length would collide into
|
|
214
|
-
one false comparable group
|
|
214
|
+
one false comparable group.
|
|
215
215
|
"""
|
|
216
216
|
if not example_hashes or any(h == "" for h in example_hashes):
|
|
217
217
|
return ""
|
|
@@ -269,7 +269,7 @@ def compute_metric_hash(metric_fn: Any) -> str:
|
|
|
269
269
|
not move when the source gains a comment or reflows whitespace; and it is
|
|
270
270
|
never empty. This replaces an earlier ``inspect.getsource`` hash that
|
|
271
271
|
produced a *different* value per environment (source vs. qualname fallback)
|
|
272
|
-
and per cosmetic edit, splitting families that should compare
|
|
272
|
+
and per cosmetic edit, splitting families that should compare.
|
|
273
273
|
Editing a metric's logic without renaming keeps the hash — intended: runs of
|
|
274
274
|
"the same metric" stay grouped.
|
|
275
275
|
"""
|
|
@@ -143,7 +143,7 @@ def _safe_serialize(value: Any, _stack: set[int] | None = None) -> Any:
|
|
|
143
143
|
from . import configuration # noqa: E402
|
|
144
144
|
|
|
145
145
|
_MAX_FIELD_JSON_SIZE = 65536 # 64 KB — a whole-payload cap, not just per-leaf
|
|
146
|
-
_MAX_DICT_KEYS = 50 # nested-dict key ceiling
|
|
146
|
+
_MAX_DICT_KEYS = 50 # nested-dict key ceiling; root exempt from the ceiling
|
|
147
147
|
|
|
148
148
|
|
|
149
149
|
def truncate_field(value: Any, max_size: int = _MAX_FIELD_JSON_SIZE, depth: int = 0) -> Any:
|
|
@@ -154,7 +154,7 @@ def truncate_field(value: Any, max_size: int = _MAX_FIELD_JSON_SIZE, depth: int
|
|
|
154
154
|
leaves become ``{"_truncated": True, ...}`` envelopes. Lives here (not in
|
|
155
155
|
``helpers``) because ``helpers`` imports ``models``; ``helpers.truncate_field``
|
|
156
156
|
delegates to this so both the callback capture path and ``SpanBuilder.to_dict``
|
|
157
|
-
share one truncator
|
|
157
|
+
share one truncator. ``max_size`` is a whole-payload bound: the
|
|
158
158
|
returned value serializes within ``max_size`` at the top level (depth 0), not
|
|
159
159
|
merely per leaf. ``depth`` exempts the root ``inputs``/``outputs`` dict from the
|
|
160
160
|
nested-dict key ceiling (it still obeys the whole-payload bound — see
|
|
@@ -167,7 +167,7 @@ def truncate_field(value: Any, max_size: int = _MAX_FIELD_JSON_SIZE, depth: int
|
|
|
167
167
|
# branches and the exporter use. json.dumps quotes and (ensure_ascii
|
|
168
168
|
# default) escapes non-ASCII to \uXXXX, matching the exporter's wire
|
|
169
169
|
# bytes (exporter.py); a raw char count or UTF-8 byte count both
|
|
170
|
-
# undercount that
|
|
170
|
+
# undercount that. _size_chars stays the char count.
|
|
171
171
|
if len(json.dumps(value)) > max_size:
|
|
172
172
|
return {
|
|
173
173
|
"_truncated": True,
|
|
@@ -187,7 +187,7 @@ def truncate_field(value: Any, max_size: int = _MAX_FIELD_JSON_SIZE, depth: int
|
|
|
187
187
|
# Show the first up-to-10 items, each recursively truncated to its own
|
|
188
188
|
# slice of the budget. This guarantees >=1 item always survives — an
|
|
189
189
|
# oversized first item becomes its own nested truncated envelope rather
|
|
190
|
-
# than breaking the loop with an empty preview
|
|
190
|
+
# than breaking the loop with an empty preview — while total
|
|
191
191
|
# preview stays bounded to ~max_size. _shown_items equals len(_preview).
|
|
192
192
|
shown = [truncate_field(item, max_size // 10, depth + 1) for item in value[:10]]
|
|
193
193
|
# Whole-payload guard: drop trailing preview items until the emitted
|
|
@@ -266,7 +266,7 @@ def _truncate_dict(value: dict[Any, Any], max_size: int, depth: int) -> dict[Any
|
|
|
266
266
|
ceiling, and a whole-payload size guard.
|
|
267
267
|
|
|
268
268
|
Never drops a key to save a sibling's bytes (a missing key reads as "data
|
|
269
|
-
wasn't captured"
|
|
269
|
+
wasn't captured" whereas a truncated value reads as
|
|
270
270
|
expected): keeps small values whole and splits the remaining budget across
|
|
271
271
|
the large ones. Two things can still drop keys, both recorded via a
|
|
272
272
|
``_type:"dict"`` envelope (``_total_keys``/``_shown_keys``, kept keys nested
|
|
@@ -534,7 +534,7 @@ class TraceBuilder:
|
|
|
534
534
|
span_count: int = 0
|
|
535
535
|
|
|
536
536
|
# Run associations - link traces to eval and optimization runs.
|
|
537
|
-
# Prefixed text ids (`ev_…`/`op_…`), minted via cmpnd.identity
|
|
537
|
+
# Prefixed text ids (`ev_…`/`op_…`), minted via cmpnd.identity;
|
|
538
538
|
# serve stores them in text columns. trace_id stays a UUID (OTel, Family 1).
|
|
539
539
|
eval_run_id: str | None = None
|
|
540
540
|
optimization_run_id: str | None = None
|
|
@@ -546,7 +546,7 @@ class TraceBuilder:
|
|
|
546
546
|
streaming: bool = False
|
|
547
547
|
# The root module's span type (e.g. "rlm"), sent in the /start payload so
|
|
548
548
|
# serve can classify a streaming trace the moment it starts — before any
|
|
549
|
-
# span has landed — rather than inferring it from span shape later
|
|
549
|
+
# span has landed — rather than inferring it from span shape later.
|
|
550
550
|
root_span_type: str | None = None
|
|
551
551
|
_batch_spans: list["SpanBuilder"] = field(default_factory=list)
|
|
552
552
|
|
|
@@ -678,7 +678,7 @@ class EvalExampleBuilder:
|
|
|
678
678
|
|
|
679
679
|
# eval_run_id links the example to its run (`ev_…`/UUID, server-assigned,
|
|
680
680
|
# so str | None — it mirrors the run's id, which is None until eval start).
|
|
681
|
-
# example_id is NOT carried: the server mints it on ingest
|
|
681
|
+
# example_id is NOT carried: the server mints it on ingest, so
|
|
682
682
|
# the SDK never sends one. trace_id stays a UUID (OTel, Family 1).
|
|
683
683
|
eval_run_id: str | None
|
|
684
684
|
example_index: int
|
|
@@ -751,7 +751,7 @@ class EvalExampleBuilder:
|
|
|
751
751
|
class EvalRunBuilder:
|
|
752
752
|
"""Mutable builder for tracking evaluation runs."""
|
|
753
753
|
|
|
754
|
-
#
|
|
754
|
+
# Server-assigned at eval start (was SDK-minted). Stays None until
|
|
755
755
|
# export_eval_start adopts the server's value (`ev_…` on serve, UUID on
|
|
756
756
|
# observe); the SDK never mints or inspects its shape.
|
|
757
757
|
eval_run_id: str | None = None
|
|
@@ -773,7 +773,7 @@ class EvalRunBuilder:
|
|
|
773
773
|
adapter_type: str | None = None
|
|
774
774
|
|
|
775
775
|
# Optimization link (set when eval is conducted during an optimization run).
|
|
776
|
-
# Prefixed text id (`op_…`), minted via cmpnd.identity
|
|
776
|
+
# Prefixed text id (`op_…`), minted via cmpnd.identity.
|
|
777
777
|
optimization_run_id: str | None = None
|
|
778
778
|
|
|
779
779
|
# Status
|
|
@@ -868,7 +868,7 @@ class EvalRunBuilder:
|
|
|
868
868
|
self.duration_ms = int((self.end_time - self.start_time).total_seconds() * 1000)
|
|
869
869
|
|
|
870
870
|
def to_start_dict(self) -> dict[str, Any]:
|
|
871
|
-
"""Minimal 'running' header for the synchronous eval-start POST
|
|
871
|
+
"""Minimal 'running' header for the synchronous eval-start POST.
|
|
872
872
|
|
|
873
873
|
Sends the "new" sentinel — the server mints the eval_run_id and returns
|
|
874
874
|
it. Carries only what's needed to persist a running header; per-example
|
|
@@ -21,7 +21,7 @@ Flow per :func:`optimize` call:
|
|
|
21
21
|
Returns the best :class:`~cmpnd.execution.DeployedProgram` directly — call it
|
|
22
22
|
like the module you handed in. Its ``.score`` carries the best score found, so
|
|
23
23
|
``optimize()`` matches ``deploy()``'s shape (a program you call) and just also
|
|
24
|
-
tells you how good it is.
|
|
24
|
+
tells you how good it is.
|
|
25
25
|
"""
|
|
26
26
|
|
|
27
27
|
from __future__ import annotations
|
|
@@ -367,7 +367,7 @@ def _upload_bundle(client: httpx.Client, initiated: dict[str, Any], bundle: byte
|
|
|
367
367
|
"""Upload the optimization bundle to the slot the initiate step returned.
|
|
368
368
|
|
|
369
369
|
Two postures, one path — keyed on ``upload_fields`` exactly as the deploy
|
|
370
|
-
upload is
|
|
370
|
+
upload is:
|
|
371
371
|
|
|
372
372
|
- **Presigned POST** (the deploy Lambda stack): ``upload_fields`` is a
|
|
373
373
|
non-empty form the destination key is pinned by; the bundle rides as the
|