cmpnd 0.7.2__tar.gz → 0.8.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {cmpnd-0.7.2 → cmpnd-0.8.1}/PKG-INFO +4 -1
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/exporter_http.py +81 -16
- cmpnd-0.8.1/cmpnd/imaging.py +162 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/models.py +15 -1
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/optimizers/gepa_callback.py +4 -12
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/optimizers/tracker.py +37 -12
- {cmpnd-0.7.2 → cmpnd-0.8.1}/docs/sdk-instrumentation.md +12 -1
- {cmpnd-0.7.2 → cmpnd-0.8.1}/pyproject.toml +11 -1
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/e2e/test_optimize_loop.py +12 -4
- cmpnd-0.8.1/tests/test_candidate_lineage.py +127 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_exporter.py +168 -5
- cmpnd-0.8.1/tests/test_imaging.py +193 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/uv.lock +102 -3
- {cmpnd-0.7.2 → cmpnd-0.8.1}/.gitignore +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/CLAUDE.md +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/README.md +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/__init__.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/_gepa_patch.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/_program_patch.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/_rlm_patch.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/callback.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/cli/__init__.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/cli/api_client.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/cli/commands/__init__.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/cli/commands/admin.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/cli/commands/auth.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/cli/commands/datasets.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/cli/commands/deployments.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/cli/commands/evals.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/cli/commands/login.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/cli/commands/optimizations.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/cli/commands/org.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/cli/commands/programs.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/cli/commands/top.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/cli/commands/traces.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/cli/credentials.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/cli/shell.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/cli/trace_render.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/cli/verbs.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/configuration.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/context.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/dataset_sync.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/datasets.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/decorators.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/deployment.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/eval_handler.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/execution.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/exporter.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/helpers.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/identity.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/ir/__init__.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/ir/decode.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/ir/encode.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/ir/hash.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/ir/program.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/ir/reflection.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/ir/types.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/ir/xxh64.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/optimization.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/optimizers/__init__.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/optimizers/gepa.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/optimizers/resume.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/cmpnd/packaging.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/docs/2026-04-02-code-review.md +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/docs/plans/2026-04-05-train-val-dataset-capture-design.md +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/docs/plans/2026-04-05-train-val-dataset-capture.md +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/examples/local_ollama.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/scripts/capture_review_run.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/scripts/fixtures/review_run.json.gz +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/scripts/gepa_resume_bug.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/scripts/seed_review_run.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/stubs/dspy/__init__.pyi +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/stubs/dspy/primitives/__init__.pyi +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/stubs/dspy/primitives/prediction.pyi +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/stubs/dspy/utils/__init__.pyi +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/stubs/dspy/utils/callback.pyi +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/__init__.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/_exporter_helpers.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/_identity_helpers.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/e2e/__init__.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/e2e/committee_task.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/e2e/conftest.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/e2e/data/committee_train.ndjson.gz +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/e2e/data/committee_val.ndjson.gz +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/e2e/fake_lm.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/e2e/helpers.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/e2e/test_capture_prompts.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/e2e/test_composite_module.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/e2e/test_dataset_auto_creation.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/e2e/test_deploy_execute.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/e2e/test_deploy_programs.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/e2e/test_edge_cases.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/e2e/test_example_hash.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/e2e/test_gepa_optimization.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/e2e/test_lm_accuracy.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/e2e/test_module_tracing.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/e2e/test_optimization_lineage.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/e2e/test_optimization_metric_identity.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/e2e/test_optimize_failures.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/e2e/test_optimize_review_committee.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/e2e/test_optimize_server_owned_completion.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/e2e/test_parallel_export.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/e2e/test_project_attribution.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/e2e/test_react_tool_spans.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/e2e/test_save_load.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/e2e/test_trace_structure.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/e2e/test_user_attribution.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/identity_vectors/01_predict_qa.json +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/identity_vectors/02_predict_renamed_field.json +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/identity_vectors/03_chain_of_thought_qa.json +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/identity_vectors/04_predict_richer_types.json +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/identity_vectors/05_composite_two_predicts.json +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/identity_vectors/06_multichaincomparison.json +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/identity_vectors/07_program_of_thought.json +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/identity_vectors/08_opaque_module.json +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/identity_vectors/09_predict_pydantic_field.json +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/identity_vectors/10_react.json +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/identity_vectors/11_refine.json +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/identity_vectors/12_retrieve.json +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/identity_vectors/13_knn.json +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/identity_vectors/14_parallel.json +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/identity_vectors/README.md +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/integration/__init__.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/integration/conftest.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/integration/test_cli_deploy_smoketest.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/integration/test_logs_renders_traces.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/integration/test_optimizing_progress.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/integration/test_unhappy_path.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/parity/__init__.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/parity/conftest.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/parity/seed_parity.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/parity/test_auth.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/parity/test_datasets.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/parity/test_deployments.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/parity/test_evals.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/parity/test_health.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/parity/test_modules.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/parity/test_optimizations.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/parity/test_programs.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/parity/test_signatures.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/parity/test_trace_streaming.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/parity/test_traces.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/stress/__init__.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/stress/_provider.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/stress/_reliability.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/stress/mint_bedrock_token.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/stress/test_deploy_programs.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/stress/test_multi_client_stress.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/stress/test_optimize_client_committee.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/stress/test_optimize_loop.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/stress/test_optimize_loop_wasm.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/stress/test_optimize_stress.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/stress/test_provider_roles.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/stress/test_real_provider_twins.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/stress/test_reliability.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/stress/test_reliability_classify.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/stress/test_rlm_wasm.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_callback.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_cli/__init__.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_cli/conftest.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_cli/test_admin.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_cli/test_api_client.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_cli/test_auth.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_cli/test_credentials.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_cli/test_datasets.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_cli/test_deployments.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_cli/test_evals.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_cli/test_login.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_cli/test_main.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_cli/test_optimizations.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_cli/test_org.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_cli/test_parity.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_cli/test_programs.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_cli/test_shell.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_cli/test_top.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_cli/test_traces.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_clustering/__init__.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_clustering/conftest.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_clustering/test_edge_cases.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_clustering/test_hash_consistency.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_clustering/test_type_conversion.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_cmp551_grouping_identity.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_configuration.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_context.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_context_propagation.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_dataset_polling.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_dataset_sync.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_datasets.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_decorators.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_deploy.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_eval.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_eval_start.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_exception_paths.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_execute.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_exporter_real.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_fake_lm.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_gepa_callback.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_gepa_integration.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_gepa_resume.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_identity.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_import_weight.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_integration.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_ir_encoder_properties.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_ir_golden_vectors.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_ir_roundtrip.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_models.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_optimization_progress.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_optimization_start.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_optimize.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_optimize_lift.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_optimize_terminal_failure.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_packaging.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_packaging_callable_source.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_packaging_program_source.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_program_lineage_persist.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_project_payload.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_reflector_properties.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_reflector_smoke.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_repro_reactv2_bugs.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_rlm_patch.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_schemas.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_seed_review_run.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_stats_export.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_trace_render.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_tracked_gepa.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_tracker_enrich.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_truncation.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/tests/test_xxh64.py +0 -0
- {cmpnd-0.7.2 → cmpnd-0.8.1}/todo.md +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: cmpnd
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.8.1
|
|
4
4
|
Summary: DSPy observability and deployment SDK for cmpnd
|
|
5
5
|
Project-URL: Homepage, https://cmpnd.ai
|
|
6
6
|
Author-email: cmpnd <hello@cmpnd.ai>
|
|
@@ -28,11 +28,14 @@ Requires-Dist: boto3>=1.34; extra == 'dev'
|
|
|
28
28
|
Requires-Dist: hypothesis>=6.100; extra == 'dev'
|
|
29
29
|
Requires-Dist: jsonschema>=4.0; extra == 'dev'
|
|
30
30
|
Requires-Dist: numpy>=1.26; extra == 'dev'
|
|
31
|
+
Requires-Dist: pillow>=10.0; extra == 'dev'
|
|
31
32
|
Requires-Dist: pytest-asyncio>=0.23; extra == 'dev'
|
|
32
33
|
Requires-Dist: pytest-rerunfailures>=14.0; extra == 'dev'
|
|
33
34
|
Requires-Dist: pytest-xdist>=3.0; extra == 'dev'
|
|
34
35
|
Requires-Dist: pytest>=8.0; extra == 'dev'
|
|
35
36
|
Requires-Dist: ruff>=0.4.0; extra == 'dev'
|
|
37
|
+
Provides-Extra: images
|
|
38
|
+
Requires-Dist: pillow>=10.0; extra == 'images'
|
|
36
39
|
Provides-Extra: stub
|
|
37
40
|
Requires-Dist: cmpnd-deploy; (python_version >= '3.11') and extra == 'stub'
|
|
38
41
|
Requires-Dist: fastapi>=0.110; extra == 'stub'
|
|
@@ -62,21 +62,68 @@ def _send_json(client: httpx.Client, method: str, path: str, payload: Any) -> ht
|
|
|
62
62
|
# so a re-POST of a request that actually landed can't double-insert.
|
|
63
63
|
_RETRYABLE_STATUS = frozenset({502, 503, 504})
|
|
64
64
|
|
|
65
|
+
# Backpressure, not unavailability: the server's buffer is full and it wants the
|
|
66
|
+
# record dropped, so the caller counts it as *dropped* and we never retry it —
|
|
67
|
+
# not even if the body carried a retryable flag. Named because that exemption is
|
|
68
|
+
# the one thing the body-flag read below must not be able to override.
|
|
69
|
+
_DROP_STATUS = 429
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _server_says_retryable(response: httpx.Response) -> bool:
|
|
73
|
+
"""Whether the server explicitly marked this error response retryable.
|
|
74
|
+
|
|
75
|
+
The `"retryable": true` error-body field is the server's stable machine
|
|
76
|
+
signal for "this failed transiently and did not commit — send it again". It
|
|
77
|
+
is what keeps this side off a hard-coded status list: the server can start
|
|
78
|
+
signalling a new transient condition without an SDK release. A body that
|
|
79
|
+
isn't JSON, isn't an object, or lacks the field reads as not-retryable.
|
|
80
|
+
"""
|
|
81
|
+
try:
|
|
82
|
+
body = response.json()
|
|
83
|
+
except Exception:
|
|
84
|
+
return False
|
|
85
|
+
return isinstance(body, dict) and body.get("retryable") is True
|
|
86
|
+
|
|
65
87
|
|
|
66
88
|
def _is_retryable_export_error(exc: BaseException) -> bool:
|
|
67
89
|
"""Whether a failed data send is a transient delivery blip worth retrying.
|
|
68
90
|
|
|
69
|
-
True for gateway/overload statuses (502/503/504)
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
91
|
+
True for gateway/overload statuses (502/503/504), a response the server
|
|
92
|
+
explicitly flagged retryable (see `_server_says_retryable`), and httpx
|
|
93
|
+
transport/timeout errors (connect/read/pool timeouts, connection
|
|
94
|
+
refused/reset). False for everything else — 4xx (client error, a retry won't
|
|
95
|
+
help), 429 (backpressure, the caller records it as *dropped* not errored, and
|
|
96
|
+
no body flag overrides that), and an unflagged 500 (the platform's to answer
|
|
97
|
+
for, mirroring the is_external_outage boundary). Non-httpx exceptions are not
|
|
98
|
+
retried (a bug in our own build path shouldn't loop)."""
|
|
75
99
|
if isinstance(exc, httpx.HTTPStatusError):
|
|
76
|
-
|
|
100
|
+
status = exc.response.status_code
|
|
101
|
+
if status == _DROP_STATUS:
|
|
102
|
+
return False
|
|
103
|
+
return status in _RETRYABLE_STATUS or _server_says_retryable(exc.response)
|
|
77
104
|
return isinstance(exc, (httpx.TimeoutException, httpx.TransportError))
|
|
78
105
|
|
|
79
106
|
|
|
107
|
+
def _retry_after_seconds(exc: BaseException) -> float | None:
|
|
108
|
+
"""The server's own backoff hint for this failure, in seconds, or None.
|
|
109
|
+
|
|
110
|
+
Only the delta-seconds form of `Retry-After` is honored — that is what the
|
|
111
|
+
server emits alongside a retryable error. A missing header, the HTTP-date
|
|
112
|
+
form, or an unparseable value returns None so the caller falls back to its
|
|
113
|
+
own exponential backoff rather than trusting a number it can't read. A
|
|
114
|
+
transport error has no response at all and so has no hint.
|
|
115
|
+
"""
|
|
116
|
+
if not isinstance(exc, httpx.HTTPStatusError):
|
|
117
|
+
return None
|
|
118
|
+
raw = exc.response.headers.get("retry-after")
|
|
119
|
+
if not raw:
|
|
120
|
+
return None
|
|
121
|
+
try:
|
|
122
|
+
return max(0.0, float(raw))
|
|
123
|
+
except ValueError:
|
|
124
|
+
return None
|
|
125
|
+
|
|
126
|
+
|
|
80
127
|
@dataclass
|
|
81
128
|
class ExportItem:
|
|
82
129
|
"""Item in the export queue."""
|
|
@@ -144,6 +191,12 @@ class BatchExporter:
|
|
|
144
191
|
# can zero the backoff. Distinct from the start-post budget above.
|
|
145
192
|
self._data_post_attempts = 3
|
|
146
193
|
self._data_post_backoff = 0.5
|
|
194
|
+
# Ceiling on a single inter-attempt wait, including one the server asked
|
|
195
|
+
# for via Retry-After. The exporter thread is the only thing draining the
|
|
196
|
+
# queue, so an over-long wait doesn't just delay this send — it stalls
|
|
197
|
+
# every queued record behind it and risks the queue filling. Honoring the
|
|
198
|
+
# server's hint is worth it; honoring an unbounded one is not.
|
|
199
|
+
self._data_post_max_delay = 5.0
|
|
147
200
|
|
|
148
201
|
# Guard against registering multiple exit handlers on repeated start()
|
|
149
202
|
self._atexit_registered = False
|
|
@@ -473,13 +526,16 @@ class BatchExporter:
|
|
|
473
526
|
def _send_json_retrying(self, client: httpx.Client, method: str, path: str, payload: Any) -> httpx.Response:
|
|
474
527
|
"""`_send_json` + `raise_for_status`, retrying transient delivery blips.
|
|
475
528
|
|
|
476
|
-
Retries `_data_post_attempts` times
|
|
477
|
-
|
|
478
|
-
`_is_retryable_export_error`)
|
|
479
|
-
|
|
480
|
-
|
|
481
|
-
|
|
482
|
-
|
|
529
|
+
Retries `_data_post_attempts` times on a retryable failure — a
|
|
530
|
+
gateway/overload status, a response the server flagged `retryable`, or a
|
|
531
|
+
transport/timeout error (see `_is_retryable_export_error`) — then
|
|
532
|
+
re-raises. The wait between attempts is the server's `Retry-After` hint
|
|
533
|
+
when it sent one, else exponential backoff, capped either way by
|
|
534
|
+
`_data_post_max_delay`. A non-retryable failure (4xx, 429, an unflagged
|
|
535
|
+
500, or a 2xx that somehow raised) propagates on the first attempt, so
|
|
536
|
+
the caller's existing except blocks still record it (429 → dropped,
|
|
537
|
+
everything else → error) exactly as before — the only change is that a
|
|
538
|
+
transient blip now gets a few tries before it counts."""
|
|
483
539
|
last_exc: Exception | None = None
|
|
484
540
|
for attempt in range(self._data_post_attempts):
|
|
485
541
|
try:
|
|
@@ -489,10 +545,19 @@ class BatchExporter:
|
|
|
489
545
|
except Exception as e:
|
|
490
546
|
last_exc = e
|
|
491
547
|
if _is_retryable_export_error(e) and attempt + 1 < self._data_post_attempts:
|
|
548
|
+
delay = _retry_after_seconds(e)
|
|
549
|
+
if delay is None:
|
|
550
|
+
delay = self._data_post_backoff * (2**attempt)
|
|
551
|
+
delay = min(delay, self._data_post_max_delay)
|
|
492
552
|
logger.debug(
|
|
493
|
-
"cmpnd: retrying %s %s after transient error (attempt %d): %s",
|
|
553
|
+
"cmpnd: retrying %s %s after transient error in %.2fs (attempt %d): %s",
|
|
554
|
+
method,
|
|
555
|
+
path,
|
|
556
|
+
delay,
|
|
557
|
+
attempt + 1,
|
|
558
|
+
e,
|
|
494
559
|
)
|
|
495
|
-
time.sleep(
|
|
560
|
+
time.sleep(delay)
|
|
496
561
|
continue
|
|
497
562
|
raise
|
|
498
563
|
assert last_exc is not None # unreachable: the loop either returns or raises
|
|
@@ -0,0 +1,162 @@
|
|
|
1
|
+
"""Optional image thumbnailing for oversized image inputs/outputs.
|
|
2
|
+
|
|
3
|
+
When a captured field is an oversized inline image data URI, the truncator would
|
|
4
|
+
otherwise discard every pixel and keep a 1000-char base64 fragment. If Pillow is
|
|
5
|
+
installed (`cmpnd[images]`), `image_thumbnail` decodes that data URI, downscales
|
|
6
|
+
it, and returns a small complete JPEG data URI that fits well under the field
|
|
7
|
+
size cap — so the trace page can show what the image was instead of an opaque
|
|
8
|
+
fragment.
|
|
9
|
+
|
|
10
|
+
Everything here is best-effort and must never raise into the capture path:
|
|
11
|
+
Pillow missing, an unparseable URI, a decode failure, or an implausibly large
|
|
12
|
+
source all return ``None``, and the caller falls back to plain-string
|
|
13
|
+
truncation. Only raster formats are handled; SVG and other non-raster or vector
|
|
14
|
+
mimes are refused (an SVG is markup, not pixels to resample).
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import base64
|
|
20
|
+
import binascii
|
|
21
|
+
import io
|
|
22
|
+
import re
|
|
23
|
+
|
|
24
|
+
_THUMBNAIL_JPEG_QUALITY = 70
|
|
25
|
+
|
|
26
|
+
# The thumbnail is sized to a BYTE budget, not a fixed pixel dimension: it should
|
|
27
|
+
# be as large as fits the field's own size budget (the caller passes its max_size)
|
|
28
|
+
# so the preview is as detailed as the truncation system already allows. Starting
|
|
29
|
+
# from the source's NATIVE size, it encodes and — if that overflows the budget —
|
|
30
|
+
# halves the longest edge and re-encodes, until the data URI fits. So an image
|
|
31
|
+
# already small enough is returned at full resolution with no resize, and a large
|
|
32
|
+
# one is scaled down just enough to fit. No pixel dimension is hard-coded anywhere.
|
|
33
|
+
|
|
34
|
+
# Bytes reserved for the envelope around _preview ({"_truncated":…,"_type":"image",
|
|
35
|
+
# "_size_chars":<N>,"_preview":"<uri>"}) so the whole envelope, not just the URI,
|
|
36
|
+
# serializes within the budget. Generous — covers the keys, a large _size_chars,
|
|
37
|
+
# and JSON punctuation.
|
|
38
|
+
_ENVELOPE_HEADROOM = 160
|
|
39
|
+
|
|
40
|
+
# Fallback budget when a caller doesn't pass one. Mirrors models._MAX_FIELD_JSON_SIZE
|
|
41
|
+
# (not imported: models imports this module, so importing it back would cycle);
|
|
42
|
+
# models.truncate_field passes its actual max_size, which is the real bound.
|
|
43
|
+
_DEFAULT_FIELD_BUDGET = 65536
|
|
44
|
+
|
|
45
|
+
# Refuse sources larger than this before decoding: a bound on the synchronous
|
|
46
|
+
# capture-path work, and a first guard against decompression bombs (the pixel
|
|
47
|
+
# ceiling below is the second).
|
|
48
|
+
_MAX_DECODE_BYTES = 8 * 1024 * 1024
|
|
49
|
+
|
|
50
|
+
# A conservative pixel ceiling so a small compressed payload can't expand into a
|
|
51
|
+
# huge bitmap on decode (decompression bomb). Enforced strictly against the
|
|
52
|
+
# declared image size read from the header BEFORE any pixel decode, so an
|
|
53
|
+
# oversized image is rejected outright rather than merely warned about. We never
|
|
54
|
+
# read or write Pillow's process-global Image.MAX_IMAGE_PIXELS, so this ceiling
|
|
55
|
+
# holds regardless of the host's Pillow configuration and never perturbs it.
|
|
56
|
+
_MAX_IMAGE_PIXELS = 24_000_000
|
|
57
|
+
|
|
58
|
+
# Raster mimes we thumbnail. SVG and other vector/non-raster mimes are excluded
|
|
59
|
+
# (an SVG is markup, not pixels to resample).
|
|
60
|
+
_RASTER_MIMES = ("png", "jpeg", "jpg", "gif", "webp", "bmp")
|
|
61
|
+
|
|
62
|
+
# Finds the first embedded raster image data URI and captures its base64 body
|
|
63
|
+
# (up to the first non-base64 char — e.g. the closing quote in JSON). The URI may
|
|
64
|
+
# be bare, or nested inside a wrapper: a captured field is rarely a bare data URI.
|
|
65
|
+
# DSPy serializes a dspy.Image inside custom-type markers as a JSON image_url
|
|
66
|
+
# block — `<<CUSTOM-TYPE-START-IDENTIFIER>>[{"type":"image_url","image_url":
|
|
67
|
+
# {"url":"data:image/…"}}]<<CUSTOM-TYPE-END-IDENTIFIER>>` — so the data URI sits
|
|
68
|
+
# inside an `image_url.url`. Searching handles that, a plain JSON blob, and the
|
|
69
|
+
# bare case alike.
|
|
70
|
+
_DATA_IMAGE_RE = re.compile(r"data:image/(?:" + "|".join(_RASTER_MIMES) + r");base64,([A-Za-z0-9+/]+=*)")
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _extract_image_bytes(value: str) -> bytes | None:
|
|
74
|
+
"""Return the decoded bytes of the first embedded raster-image data URI in
|
|
75
|
+
``value``, or None if there is none or it exceeds the size bound.
|
|
76
|
+
"""
|
|
77
|
+
match = _DATA_IMAGE_RE.search(value)
|
|
78
|
+
if match is None:
|
|
79
|
+
return None
|
|
80
|
+
b64 = match.group(1)
|
|
81
|
+
# Cheap size gate on the encoded text before decoding (base64 is ~4/3 the
|
|
82
|
+
# raw size); the decoded check below is the exact one.
|
|
83
|
+
if len(b64) > (_MAX_DECODE_BYTES // 3) * 4 + 4:
|
|
84
|
+
return None
|
|
85
|
+
try:
|
|
86
|
+
raw = base64.b64decode(b64, validate=True)
|
|
87
|
+
except (binascii.Error, ValueError):
|
|
88
|
+
return None
|
|
89
|
+
if len(raw) > _MAX_DECODE_BYTES:
|
|
90
|
+
return None
|
|
91
|
+
return raw
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _encode_jpeg_data_uri(img) -> str:
|
|
95
|
+
"""Encode an RGB image as a `data:image/jpeg;base64,…` string."""
|
|
96
|
+
buf = io.BytesIO()
|
|
97
|
+
img.save(buf, format="JPEG", quality=_THUMBNAIL_JPEG_QUALITY, optimize=True)
|
|
98
|
+
return "data:image/jpeg;base64," + base64.b64encode(buf.getvalue()).decode("ascii")
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def image_thumbnail(value: str, budget: int = _DEFAULT_FIELD_BUDGET) -> str | None:
|
|
102
|
+
"""Re-encode an oversized inline image as the largest JPEG data URI that fits
|
|
103
|
+
``budget`` bytes (the caller's field size cap).
|
|
104
|
+
|
|
105
|
+
Sizes by BYTES, not pixels: it encodes at the source's native size and, while
|
|
106
|
+
that overflows ``budget`` minus the envelope overhead, halves the longest edge
|
|
107
|
+
and re-encodes until it fits — so an already-small image is returned at full
|
|
108
|
+
resolution and a large one is scaled down just enough. Returns a
|
|
109
|
+
`data:image/jpeg;base64,…` string on success, or None when the value contains
|
|
110
|
+
no raster image data URI, Pillow is unavailable, no size fits the budget, or
|
|
111
|
+
anything fails. Never raises.
|
|
112
|
+
"""
|
|
113
|
+
raw = _extract_image_bytes(value)
|
|
114
|
+
if raw is None:
|
|
115
|
+
return None
|
|
116
|
+
limit = budget - _ENVELOPE_HEADROOM
|
|
117
|
+
if limit <= len("data:image/jpeg;base64,"):
|
|
118
|
+
return None
|
|
119
|
+
try:
|
|
120
|
+
from PIL import Image # optional dependency (cmpnd[images])
|
|
121
|
+
except ImportError:
|
|
122
|
+
return None
|
|
123
|
+
try:
|
|
124
|
+
with Image.open(io.BytesIO(raw)) as opened:
|
|
125
|
+
# Image.open is lazy: it reads only the header (enough for .size),
|
|
126
|
+
# not the pixels. Reject on the declared size BEFORE load() so a
|
|
127
|
+
# small compressed payload that claims a huge bitmap never decodes at
|
|
128
|
+
# full resolution. This is our own strict ceiling; we never touch
|
|
129
|
+
# Pillow's Image.MAX_IMAGE_PIXELS global.
|
|
130
|
+
w, h = opened.size
|
|
131
|
+
if w * h > _MAX_IMAGE_PIXELS:
|
|
132
|
+
return None
|
|
133
|
+
opened.load()
|
|
134
|
+
# JPEG has no alpha: flatten transparency / palette onto white so
|
|
135
|
+
# a PNG/GIF with an alpha channel doesn't fail the encode.
|
|
136
|
+
if opened.mode in ("RGBA", "LA", "P"):
|
|
137
|
+
rgba = opened.convert("RGBA")
|
|
138
|
+
background = Image.new("RGBA", rgba.size, (255, 255, 255, 255))
|
|
139
|
+
base = Image.alpha_composite(background, rgba).convert("RGB")
|
|
140
|
+
elif opened.mode != "RGB":
|
|
141
|
+
base = opened.convert("RGB")
|
|
142
|
+
else:
|
|
143
|
+
base = opened.copy()
|
|
144
|
+
# Encode at native size; while it overflows the budget, halve the
|
|
145
|
+
# longest edge and re-encode. Returns the first (largest) size that
|
|
146
|
+
# fits — native itself when the image is already small enough.
|
|
147
|
+
best: str | None = None
|
|
148
|
+
edge = max(base.size)
|
|
149
|
+
while edge >= 1:
|
|
150
|
+
candidate = base.copy()
|
|
151
|
+
candidate.thumbnail((edge, edge)) # only ever downscales
|
|
152
|
+
uri = _encode_jpeg_data_uri(candidate)
|
|
153
|
+
if len(uri) <= limit:
|
|
154
|
+
best = uri
|
|
155
|
+
break
|
|
156
|
+
edge //= 2
|
|
157
|
+
except Exception:
|
|
158
|
+
# Any decode/resize/encode failure — corrupt bytes or a header that
|
|
159
|
+
# can't be read, an unsupported sub-format — degrades to no thumbnail.
|
|
160
|
+
# Instrumentation must not break the host program.
|
|
161
|
+
return None
|
|
162
|
+
return best
|
|
@@ -9,7 +9,7 @@ from enum import Enum
|
|
|
9
9
|
from typing import Any
|
|
10
10
|
from uuid import UUID
|
|
11
11
|
|
|
12
|
-
from . import identity
|
|
12
|
+
from . import identity, imaging
|
|
13
13
|
|
|
14
14
|
|
|
15
15
|
def _safe_serialize(value: Any, _stack: set[int] | None = None) -> Any:
|
|
@@ -169,6 +169,20 @@ def truncate_field(value: Any, max_size: int = _MAX_FIELD_JSON_SIZE, depth: int
|
|
|
169
169
|
# bytes (exporter.py); a raw char count or UTF-8 byte count both
|
|
170
170
|
# undercount that. _size_chars stays the char count.
|
|
171
171
|
if len(json.dumps(value)) > max_size:
|
|
172
|
+
# An oversized inline image would otherwise lose every pixel to the
|
|
173
|
+
# 1000-char fragment below. If the value is a raster image data URI
|
|
174
|
+
# and Pillow is available, keep a small COMPLETE thumbnail in the
|
|
175
|
+
# _preview slot and flip _type to "image" so the reader can render it.
|
|
176
|
+
# Returns None (→ plain-string truncation) for non-images, no Pillow,
|
|
177
|
+
# or any failure, so this path is purely additive and never raises.
|
|
178
|
+
thumbnail = imaging.image_thumbnail(value, max_size)
|
|
179
|
+
if thumbnail is not None:
|
|
180
|
+
return {
|
|
181
|
+
"_truncated": True,
|
|
182
|
+
"_type": "image",
|
|
183
|
+
"_size_chars": len(value),
|
|
184
|
+
"_preview": thumbnail,
|
|
185
|
+
}
|
|
172
186
|
return {
|
|
173
187
|
"_truncated": True,
|
|
174
188
|
"_type": "string",
|
|
@@ -464,17 +464,9 @@ class CmpndGEPACallback:
|
|
|
464
464
|
|
|
465
465
|
for idx, instructions in enumerate(candidates):
|
|
466
466
|
try:
|
|
467
|
-
# Extract parent
|
|
468
|
-
|
|
469
|
-
if idx < len(parents)
|
|
470
|
-
parent_list = parents[idx]
|
|
471
|
-
if isinstance(parent_list, (list, tuple)):
|
|
472
|
-
for p in parent_list:
|
|
473
|
-
if p is not None:
|
|
474
|
-
parent_idx = p
|
|
475
|
-
break
|
|
476
|
-
elif parent_list is not None:
|
|
477
|
-
parent_idx = parent_list
|
|
467
|
+
# Extract the full parent list — a merge proposal has two, and
|
|
468
|
+
# serve derives the mutation type and generation from them.
|
|
469
|
+
parents_of = tracker.normalize_parent_indices(parents[idx] if idx < len(parents) else None)
|
|
478
470
|
|
|
479
471
|
# Compute aggregate score: prefer val_subscores (per-example),
|
|
480
472
|
# fall back to objective_scores (per-objective)
|
|
@@ -503,7 +495,7 @@ class CmpndGEPACallback:
|
|
|
503
495
|
tracker.CandidateData(
|
|
504
496
|
candidate_idx=idx,
|
|
505
497
|
instructions=instructions if isinstance(instructions, dict) else {},
|
|
506
|
-
|
|
498
|
+
parent_indices=parents_of,
|
|
507
499
|
discovery_iteration=int(discovery_counts[idx]) if idx < len(discovery_counts) else 0,
|
|
508
500
|
val_aggregate_score=val_aggregate_score,
|
|
509
501
|
val_subscores=val_subs,
|
|
@@ -22,6 +22,25 @@ _ETA_WINDOW = 10
|
|
|
22
22
|
_ETA_MIN_SAMPLES = 3
|
|
23
23
|
|
|
24
24
|
|
|
25
|
+
def normalize_parent_indices(parent_entry: Any) -> list[int]:
|
|
26
|
+
"""Normalize one candidate's optimizer parent entry into candidate indices.
|
|
27
|
+
|
|
28
|
+
GEPA records parents per candidate as a list — two entries for a merge
|
|
29
|
+
proposal, `[None]` for the seed — while some result shapes give a bare
|
|
30
|
+
index or nothing at all. `None` entries are padding, not edges, so they
|
|
31
|
+
are dropped rather than becoming a parentless-but-present link.
|
|
32
|
+
"""
|
|
33
|
+
if parent_entry is None:
|
|
34
|
+
return []
|
|
35
|
+
if isinstance(parent_entry, int):
|
|
36
|
+
return [int(parent_entry)]
|
|
37
|
+
if isinstance(parent_entry, (list, tuple, set, frozenset)):
|
|
38
|
+
entries: list[Any] = list(parent_entry)
|
|
39
|
+
return [int(p) for p in entries if p is not None]
|
|
40
|
+
logger.debug(f"Unknown parent format: {type(parent_entry)} = {parent_entry}")
|
|
41
|
+
return []
|
|
42
|
+
|
|
43
|
+
|
|
25
44
|
def _sanitize_for_json(obj: Any) -> Any:
|
|
26
45
|
"""Recursively sanitize an object for JSON serialization.
|
|
27
46
|
|
|
@@ -155,7 +174,14 @@ class CandidateData:
|
|
|
155
174
|
|
|
156
175
|
candidate_idx: int
|
|
157
176
|
instructions: dict[str, str] # component_name -> instruction text
|
|
177
|
+
# Lineage. `parent_indices` is the full parent list — a GEPA merge proposal
|
|
178
|
+
# has two, and serve derives mutation_type (seed / reflective / merge) and
|
|
179
|
+
# generation from it, so collapsing to one parent loses the merge. Indices
|
|
180
|
+
# are into this run's candidate list: serve mints the candidate ids, so an
|
|
181
|
+
# index is the only lineage the SDK can name. `parent_idx` is the first
|
|
182
|
+
# parent, kept for the single-parent call sites and read surfaces.
|
|
158
183
|
parent_idx: int | None = None
|
|
184
|
+
parent_indices: list[int] = field(default_factory=list)
|
|
159
185
|
discovery_iteration: int = 0
|
|
160
186
|
|
|
161
187
|
# Validation scores
|
|
@@ -174,12 +200,20 @@ class CandidateData:
|
|
|
174
200
|
subsample_score_before: float | None = None # parent's subsample score
|
|
175
201
|
subsample_score_after: float | None = None # this candidate's subsample score
|
|
176
202
|
|
|
203
|
+
def __post_init__(self) -> None:
|
|
204
|
+
"""Reconcile the two parent forms so both are always populated."""
|
|
205
|
+
self.parent_indices = normalize_parent_indices(self.parent_indices)
|
|
206
|
+
if not self.parent_indices and self.parent_idx is not None:
|
|
207
|
+
self.parent_indices = [int(self.parent_idx)]
|
|
208
|
+
self.parent_idx = self.parent_indices[0] if self.parent_indices else None
|
|
209
|
+
|
|
177
210
|
def to_api_dict(self) -> dict[str, Any]:
|
|
178
211
|
"""Convert to API-compatible dictionary."""
|
|
179
212
|
return {
|
|
180
213
|
"candidate_idx": int(self.candidate_idx),
|
|
181
214
|
"instructions": _sanitize_for_json(self.instructions),
|
|
182
215
|
"parent_idx": int(self.parent_idx) if self.parent_idx is not None else None,
|
|
216
|
+
"parent_indices": [int(p) for p in self.parent_indices],
|
|
183
217
|
"discovery_iteration": int(self.discovery_iteration) if self.discovery_iteration else 0,
|
|
184
218
|
"val_aggregate_score": float(self.val_aggregate_score) if self.val_aggregate_score else 0.0,
|
|
185
219
|
"val_subscores": [float(s) for s in self.val_subscores] if self.val_subscores else [],
|
|
@@ -563,24 +597,15 @@ class OptimizationTracker:
|
|
|
563
597
|
elif isinstance(candidate, dict):
|
|
564
598
|
instructions = candidate
|
|
565
599
|
|
|
566
|
-
|
|
567
|
-
if idx < len(parents_list):
|
|
568
|
-
parent_val = parents_list[idx]
|
|
569
|
-
# Handle different parent structures
|
|
570
|
-
if isinstance(parent_val, (list, tuple)) and len(parent_val) > 0:
|
|
571
|
-
parent_idx = parent_val[0]
|
|
572
|
-
elif isinstance(parent_val, int):
|
|
573
|
-
parent_idx = parent_val
|
|
574
|
-
elif parent_val is not None:
|
|
575
|
-
logger.debug(f"Unknown parent format for candidate {idx}: {type(parent_val)} = {parent_val}")
|
|
600
|
+
parents = normalize_parent_indices(parents_list[idx] if idx < len(parents_list) else None)
|
|
576
601
|
|
|
577
|
-
logger.debug(f"Candidate {idx}:
|
|
602
|
+
logger.debug(f"Candidate {idx}: parents={parents}")
|
|
578
603
|
|
|
579
604
|
self.candidates.append(
|
|
580
605
|
CandidateData(
|
|
581
606
|
candidate_idx=idx,
|
|
582
607
|
instructions=instructions,
|
|
583
|
-
|
|
608
|
+
parent_indices=parents,
|
|
584
609
|
discovery_iteration=discovery_counts[idx] if idx < len(discovery_counts) else 0,
|
|
585
610
|
val_aggregate_score=scores_list[idx] if idx < len(scores_list) else 0.0,
|
|
586
611
|
val_subscores=subscores_list[idx] if idx < len(subscores_list) else [],
|
|
@@ -156,7 +156,17 @@ Override any of these by passing kwargs to `cmpnd.configure(...)`.
|
|
|
156
156
|
|
|
157
157
|
### Delivery errors, dropping, and transient retry
|
|
158
158
|
|
|
159
|
-
Each `_send_*` method routes a failed delivery to one of two counters: a `429` (server buffer full) increments `_dropped_count` and is *not* retried (it's backpressure — the caller intentionally sheds the load), while any other non-2xx or transport failure increments `_error_count`. The data POSTs go through `_send_json_retrying` (`exporter_http.py`), which retries a **transient** failure
|
|
159
|
+
Each `_send_*` method routes a failed delivery to one of two counters: a `429` (server buffer full) increments `_dropped_count` and is *not* retried (it's backpressure — the caller intentionally sheds the load), while any other non-2xx or transport failure increments `_error_count`. The data POSTs go through `_send_json_retrying` (`exporter_http.py`), which retries a **transient** failure up to `_data_post_attempts` (3) times before it counts, so a momentary blip against a busy backend doesn't register as a hard error. Three things count as transient, checked by `_is_retryable_export_error`:
|
|
160
|
+
|
|
161
|
+
- a `502`/`503`/`504` gateway/overload status;
|
|
162
|
+
- **any** status the server explicitly flagged with a `{"retryable": true}` error body (`_server_says_retryable`) — the server's stable machine signal for "this failed transiently and did not commit", which is what keeps this side off a hard-coded status list and lets the server start signalling a new transient condition without an SDK release;
|
|
163
|
+
- an httpx transport/timeout error.
|
|
164
|
+
|
|
165
|
+
It is *not* retried on `4xx` (client error), `429`, or an **unflagged** `500` (the platform's to answer for — the same boundary as the stress harness's `is_external_outage`). The `429` exemption is absolute: a flagged `429` still drops, because retrying into a server that asked us to shed would amplify the overload the status exists to relieve.
|
|
166
|
+
|
|
167
|
+
The wait between attempts is the server's `Retry-After` hint when it sent one (delta-seconds form only; an HTTP-date or unparseable value falls back), else `_data_post_backoff` exponential backoff. Either way it is clamped to `_data_post_max_delay` (5s) — the exporter thread is the sole queue drainer, so an unbounded wait wouldn't just delay one send, it would stall every record behind it.
|
|
168
|
+
|
|
169
|
+
Retrying is safe because serve dedups ingest — traces on `(org_id, trace_id, start_time)`, spans on the `spans_dedup` index — so a re-POST of a request that actually landed can't double-insert; on the flagged path the server only sets the flag when it did *not* commit. The blocking run-*start* POSTs keep their own separate retry budget (`_start_post_attempts`), since they must return a server-assigned id synchronously.
|
|
160
170
|
|
|
161
171
|
### Loose spans vs bundled traces
|
|
162
172
|
|
|
@@ -245,6 +255,7 @@ Every envelope carries `_truncated: true`, a `_type`, and a `_preview` (serve re
|
|
|
245
255
|
- **list** → `{_truncated, _type: "list", _total_items, _shown_items, _size_bytes, _preview}` — `_preview` is the first ≤10 items, each recursively truncated (`_shown_items == len(_preview)`); an oversized first item survives as its own nested envelope.
|
|
246
256
|
- **dict** (nested, over the key ceiling `_MAX_DICT_KEYS` = 50) → `{_truncated, _type: "dict", _total_keys, _shown_keys, _preview}` — `_preview` holds the first K kept keys, each recursively truncated. Kept keys nest under `_preview` (not inline) so a user key named like a marker can't be clobbered.
|
|
247
257
|
- **other scalar** → `{_truncated, _type: <typename>, _size_bytes | _size_chars, _preview}`.
|
|
258
|
+
- **image** → `{_truncated, _type: "image", _size_chars, _preview}` — when an oversized string is a raster image data URI (`data:image/<png|jpeg|gif|webp|bmp>;base64,…`) **and** the optional `cmpnd[images]` extra (Pillow) is installed, `_preview` holds a *complete* small JPEG thumbnail — sized to fit the field's byte budget (as large as fits, scaled down from the source's native size, with no fixed pixel dimension) — instead of a 1000-char fragment, and `_type` is `"image"`. Emitted only on a successful thumbnail; without Pillow, or on any failure, the field degrades to the plain **string** envelope above (byte-identical to before). The `_type` is the render gate: serve shows `_preview` as an `<img>` only for `_type:"image"` (a `string`-envelope fragment also starts with `data:image/` but is incomplete, so it stays text). Thumbnailing lives in `python-sdk/cmpnd/imaging.py`; it is import-guarded and never raises into the capture path.
|
|
248
259
|
|
|
249
260
|
Dict handling: **value-aware allocation** keeps every key (small values whole, the remaining budget split across the large ones — a missing key would read as "not captured", the symptom this replaced). Two things can still drop keys, each recorded via the `_type:"dict"` envelope: the **key ceiling** (`_MAX_DICT_KEYS`), which applies to nested dicts only — the root is exempt from it; and the **whole-payload guard** — since `max_size` bounds the whole payload, after assembly any dict (root included) sheds trailing keys until it serializes within `max_size`. So the root keeps all its fields when they fit and is trimmed (with `_total_keys`/`_shown_keys`) when the payload still exceeds the cap. The callback capture path (`helpers.clean_inputs`) truncates each field value at depth 1, so a nested dict field value caps consistently with `SpanBuilder.to_dict` (where the whole inputs dict is the depth-0 root and field values are depth 1).
|
|
250
261
|
|
|
@@ -5,7 +5,7 @@ name = "cmpnd"
|
|
|
5
5
|
# it's new) — no git tags. Bump with `uv version --bump {minor,patch}` or by
|
|
6
6
|
# hand. Runtime `cmpnd.__version__` reads installed metadata, so it tracks
|
|
7
7
|
# this automatically. See docs/plans/2026-07-02-sdk-pypi-tag-free-publish.md.
|
|
8
|
-
version = "0.
|
|
8
|
+
version = "0.8.1"
|
|
9
9
|
description = "DSPy observability and deployment SDK for cmpnd"
|
|
10
10
|
readme = "README.md"
|
|
11
11
|
requires-python = ">=3.10"
|
|
@@ -46,6 +46,13 @@ dependencies = [
|
|
|
46
46
|
]
|
|
47
47
|
|
|
48
48
|
[project.optional-dependencies]
|
|
49
|
+
# Image thumbnailing for oversized image inputs/outputs. Optional and
|
|
50
|
+
# import-guarded: without it, an oversized image degrades to today's text
|
|
51
|
+
# preview. Pillow is heavyweight and most instrumentation captures no images, so
|
|
52
|
+
# it is not a hard dependency.
|
|
53
|
+
images = [
|
|
54
|
+
"pillow>=10.0",
|
|
55
|
+
]
|
|
49
56
|
dev = [
|
|
50
57
|
"pytest>=8.0",
|
|
51
58
|
"pytest-asyncio>=0.23",
|
|
@@ -62,6 +69,9 @@ dev = [
|
|
|
62
69
|
# unconditionally so it rode in transitively; dspy 3.3.0 moved numpy behind a
|
|
63
70
|
# `numpy` extra, so declare it directly as the test dependency it always was.
|
|
64
71
|
"numpy>=1.26",
|
|
72
|
+
# Exercises the image-thumbnail path (the `images` extra) so the positive
|
|
73
|
+
# branch is covered in CI, not just the Pillow-absent fallback.
|
|
74
|
+
"pillow>=10.0",
|
|
65
75
|
]
|
|
66
76
|
# Deps for `deploy/deploy/stub/`, the local FastAPI server that emulates
|
|
67
77
|
# the AWS deploy stack. The stub itself ships in the cmpnd-deploy
|
|
@@ -215,10 +215,18 @@ class TestOptimizeGepaDeployDetail:
|
|
|
215
215
|
|
|
216
216
|
# The producer ran a cmpnd OptimizationTracker: the run lands full detail
|
|
217
217
|
# (>= the seed candidate), not just a summary header. Asserted first so a
|
|
218
|
-
# producer/tracker failure is distinguished from a trace-stamping one.
|
|
219
|
-
#
|
|
220
|
-
#
|
|
221
|
-
|
|
218
|
+
# producer/tracker failure is distinguished from a trace-stamping one.
|
|
219
|
+
#
|
|
220
|
+
# Polled, because a run is addressable BEFORE it has any detail: serve
|
|
221
|
+
# writes the header at initiate with status "running" so the run is listed
|
|
222
|
+
# and its detail URL resolves for its whole duration (CMP-412), and the
|
|
223
|
+
# candidates arrive later, when the completion watcher settles it. The
|
|
224
|
+
# search above therefore returns the run while the candidate set is still
|
|
225
|
+
# legitimately empty, and a single read here races the watcher.
|
|
226
|
+
def _find_candidates():
|
|
227
|
+
return server_client.get_optimization_candidates(run_id).get("candidates", [])
|
|
228
|
+
|
|
229
|
+
candidates = _poll(_find_candidates)
|
|
222
230
|
assert candidates, f"GEPA deploy run {run_id} landed a header but no candidate detail (CMP-323)"
|
|
223
231
|
|
|
224
232
|
# CMP-323/CMP-318 activation: the Lambda's tracker stamps the run id onto
|