cmpnd 0.8.0__tar.gz → 0.8.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {cmpnd-0.8.0 → cmpnd-0.8.2}/PKG-INFO +1 -1
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/cli/commands/top.py +1 -2
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/dataset_sync.py +11 -3
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/deployment.py +19 -8
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/exporter_http.py +81 -16
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/models.py +98 -31
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/optimizers/gepa_callback.py +9 -16
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/optimizers/tracker.py +57 -20
- {cmpnd-0.8.0 → cmpnd-0.8.2}/docs/sdk-instrumentation.md +27 -1
- {cmpnd-0.8.0 → cmpnd-0.8.2}/pyproject.toml +1 -1
- cmpnd-0.8.2/tests/test_candidate_lineage.py +127 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_deploy.py +20 -4
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_exporter.py +168 -5
- cmpnd-0.8.2/tests/test_models.py +849 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/uv.lock +88 -88
- cmpnd-0.8.0/tests/test_models.py +0 -438
- {cmpnd-0.8.0 → cmpnd-0.8.2}/.gitignore +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/CLAUDE.md +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/README.md +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/__init__.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/_gepa_patch.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/_program_patch.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/_rlm_patch.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/callback.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/cli/__init__.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/cli/api_client.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/cli/commands/__init__.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/cli/commands/admin.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/cli/commands/auth.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/cli/commands/datasets.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/cli/commands/deployments.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/cli/commands/evals.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/cli/commands/login.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/cli/commands/optimizations.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/cli/commands/org.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/cli/commands/programs.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/cli/commands/traces.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/cli/credentials.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/cli/shell.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/cli/trace_render.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/cli/verbs.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/configuration.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/context.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/datasets.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/decorators.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/eval_handler.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/execution.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/exporter.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/helpers.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/identity.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/imaging.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/ir/__init__.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/ir/decode.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/ir/encode.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/ir/hash.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/ir/program.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/ir/reflection.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/ir/types.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/ir/xxh64.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/optimization.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/optimizers/__init__.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/optimizers/gepa.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/optimizers/resume.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/cmpnd/packaging.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/docs/2026-04-02-code-review.md +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/docs/plans/2026-04-05-train-val-dataset-capture-design.md +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/docs/plans/2026-04-05-train-val-dataset-capture.md +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/examples/local_ollama.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/scripts/capture_review_run.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/scripts/fixtures/review_run.json.gz +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/scripts/gepa_resume_bug.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/scripts/seed_review_run.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/stubs/dspy/__init__.pyi +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/stubs/dspy/primitives/__init__.pyi +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/stubs/dspy/primitives/prediction.pyi +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/stubs/dspy/utils/__init__.pyi +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/stubs/dspy/utils/callback.pyi +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/__init__.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/_exporter_helpers.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/_identity_helpers.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/e2e/__init__.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/e2e/committee_task.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/e2e/conftest.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/e2e/data/committee_train.ndjson.gz +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/e2e/data/committee_val.ndjson.gz +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/e2e/fake_lm.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/e2e/helpers.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/e2e/test_capture_prompts.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/e2e/test_composite_module.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/e2e/test_dataset_auto_creation.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/e2e/test_deploy_execute.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/e2e/test_deploy_programs.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/e2e/test_edge_cases.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/e2e/test_example_hash.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/e2e/test_gepa_optimization.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/e2e/test_lm_accuracy.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/e2e/test_module_tracing.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/e2e/test_optimization_lineage.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/e2e/test_optimization_metric_identity.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/e2e/test_optimize_failures.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/e2e/test_optimize_loop.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/e2e/test_optimize_review_committee.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/e2e/test_optimize_server_owned_completion.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/e2e/test_parallel_export.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/e2e/test_project_attribution.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/e2e/test_react_tool_spans.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/e2e/test_save_load.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/e2e/test_trace_structure.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/e2e/test_user_attribution.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/identity_vectors/01_predict_qa.json +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/identity_vectors/02_predict_renamed_field.json +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/identity_vectors/03_chain_of_thought_qa.json +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/identity_vectors/04_predict_richer_types.json +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/identity_vectors/05_composite_two_predicts.json +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/identity_vectors/06_multichaincomparison.json +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/identity_vectors/07_program_of_thought.json +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/identity_vectors/08_opaque_module.json +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/identity_vectors/09_predict_pydantic_field.json +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/identity_vectors/10_react.json +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/identity_vectors/11_refine.json +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/identity_vectors/12_retrieve.json +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/identity_vectors/13_knn.json +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/identity_vectors/14_parallel.json +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/identity_vectors/README.md +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/integration/__init__.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/integration/conftest.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/integration/test_cli_deploy_smoketest.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/integration/test_logs_renders_traces.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/integration/test_optimizing_progress.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/integration/test_unhappy_path.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/parity/__init__.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/parity/conftest.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/parity/seed_parity.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/parity/test_auth.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/parity/test_datasets.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/parity/test_deployments.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/parity/test_evals.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/parity/test_health.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/parity/test_modules.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/parity/test_optimizations.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/parity/test_programs.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/parity/test_signatures.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/parity/test_trace_streaming.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/parity/test_traces.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/stress/__init__.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/stress/_provider.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/stress/_reliability.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/stress/mint_bedrock_token.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/stress/test_deploy_programs.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/stress/test_multi_client_stress.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/stress/test_optimize_client_committee.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/stress/test_optimize_loop.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/stress/test_optimize_loop_wasm.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/stress/test_optimize_stress.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/stress/test_provider_roles.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/stress/test_real_provider_twins.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/stress/test_reliability.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/stress/test_reliability_classify.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/stress/test_rlm_wasm.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_callback.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_cli/__init__.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_cli/conftest.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_cli/test_admin.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_cli/test_api_client.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_cli/test_auth.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_cli/test_credentials.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_cli/test_datasets.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_cli/test_deployments.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_cli/test_evals.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_cli/test_login.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_cli/test_main.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_cli/test_optimizations.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_cli/test_org.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_cli/test_parity.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_cli/test_programs.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_cli/test_shell.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_cli/test_top.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_cli/test_traces.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_clustering/__init__.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_clustering/conftest.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_clustering/test_edge_cases.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_clustering/test_hash_consistency.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_clustering/test_type_conversion.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_cmp551_grouping_identity.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_configuration.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_context.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_context_propagation.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_dataset_polling.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_dataset_sync.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_datasets.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_decorators.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_eval.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_eval_start.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_exception_paths.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_execute.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_exporter_real.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_fake_lm.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_gepa_callback.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_gepa_integration.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_gepa_resume.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_identity.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_imaging.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_import_weight.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_integration.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_ir_encoder_properties.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_ir_golden_vectors.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_ir_roundtrip.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_optimization_progress.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_optimization_start.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_optimize.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_optimize_lift.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_optimize_terminal_failure.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_packaging.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_packaging_callable_source.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_packaging_program_source.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_program_lineage_persist.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_project_payload.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_reflector_properties.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_reflector_smoke.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_repro_reactv2_bugs.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_rlm_patch.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_schemas.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_seed_review_run.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_stats_export.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_trace_render.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_tracked_gepa.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_tracker_enrich.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_truncation.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/tests/test_xxh64.py +0 -0
- {cmpnd-0.8.0 → cmpnd-0.8.2}/todo.md +0 -0
|
@@ -622,10 +622,9 @@ def _parse_deploy_metadata(raw: str | None) -> dict[str, Any] | None:
|
|
|
622
622
|
from ... import deployment
|
|
623
623
|
|
|
624
624
|
try:
|
|
625
|
-
deployment.
|
|
625
|
+
return deployment._prepare_deploy_metadata(value)
|
|
626
626
|
except (TypeError, ValueError) as exc:
|
|
627
627
|
raise api_client.CliError(f"--metadata is invalid: {exc}")
|
|
628
|
-
return value
|
|
629
628
|
|
|
630
629
|
|
|
631
630
|
def cmd_deploy(client: api_client.ApiClient, args: Any) -> None:
|
|
@@ -12,7 +12,7 @@ from typing import Any
|
|
|
12
12
|
|
|
13
13
|
import httpx
|
|
14
14
|
|
|
15
|
-
from . import configuration, identity
|
|
15
|
+
from . import configuration, identity, models
|
|
16
16
|
|
|
17
17
|
logger = logging.getLogger(__name__)
|
|
18
18
|
|
|
@@ -113,12 +113,20 @@ def build_dataset_payload(
|
|
|
113
113
|
inputs = {k: v for k, v in d.items() if k in input_fields}
|
|
114
114
|
expected_outputs = {k: v for k, v in d.items() if k not in input_fields}
|
|
115
115
|
|
|
116
|
+
# Hashed from the example as given, deliberately: this hash keys dataset
|
|
117
|
+
# dedup and the per-run split-membership join (train/val_example_hashes,
|
|
118
|
+
# computed elsewhere from the same unscrubbed example), so it is the
|
|
119
|
+
# example's identity and must not move with a storage concern.
|
|
116
120
|
example_hash = identity.compute_example_hash(inputs) if inputs else ""
|
|
117
121
|
|
|
118
122
|
examples.append(
|
|
119
123
|
{
|
|
120
|
-
|
|
121
|
-
|
|
124
|
+
# Both land in jsonb columns, so they go through the same
|
|
125
|
+
# serializer the trace path uses — which also makes a
|
|
126
|
+
# non-JSON-native field value (a Pydantic model, an image) exportable
|
|
127
|
+
# instead of failing the encode at send time.
|
|
128
|
+
"inputs": models._safe_serialize(inputs),
|
|
129
|
+
"expected_outputs": models._safe_serialize(expected_outputs),
|
|
122
130
|
"example_hash": example_hash,
|
|
123
131
|
}
|
|
124
132
|
)
|
|
@@ -16,7 +16,7 @@ from typing import Any
|
|
|
16
16
|
|
|
17
17
|
import httpx
|
|
18
18
|
|
|
19
|
-
from . import configuration, execution, identity, packaging
|
|
19
|
+
from . import configuration, execution, identity, models, packaging
|
|
20
20
|
|
|
21
21
|
logger = logging.getLogger(__name__)
|
|
22
22
|
|
|
@@ -27,9 +27,19 @@ logger = logging.getLogger(__name__)
|
|
|
27
27
|
_MAX_DEPLOY_METADATA_BYTES = 64 * 1024
|
|
28
28
|
|
|
29
29
|
|
|
30
|
-
def
|
|
31
|
-
"""Validate the deploy metadata blob
|
|
32
|
-
|
|
30
|
+
def _prepare_deploy_metadata(metadata: Any) -> dict[str, Any]:
|
|
31
|
+
"""Validate the deploy metadata blob and return it in storable form.
|
|
32
|
+
|
|
33
|
+
Raises at the call site on a bad value so the user sees a clear message
|
|
34
|
+
instead of a server 422/413. Validation and scrubbing are one step on
|
|
35
|
+
purpose: the returned blob lands in a jsonb column, so a caller that
|
|
36
|
+
validated without scrubbing would still be able to ship a payload the
|
|
37
|
+
column must reject.
|
|
38
|
+
|
|
39
|
+
The size cap is measured on the value as given — scrubbing only ever
|
|
40
|
+
shortens it, so the check stays the conservative one the server's own
|
|
41
|
+
limit is stated against.
|
|
42
|
+
"""
|
|
33
43
|
if not isinstance(metadata, dict):
|
|
34
44
|
raise TypeError(f"deploy(metadata=...) must be a dict, got {type(metadata).__name__}")
|
|
35
45
|
if not all(isinstance(k, str) for k in metadata):
|
|
@@ -40,6 +50,7 @@ def _validate_deploy_metadata(metadata: Any) -> None:
|
|
|
40
50
|
raise TypeError(f"deploy(metadata=...) must be JSON-serializable: {exc}") from exc
|
|
41
51
|
if len(encoded.encode("utf-8")) > _MAX_DEPLOY_METADATA_BYTES:
|
|
42
52
|
raise ValueError(f"deploy(metadata=...) exceeds the {_MAX_DEPLOY_METADATA_BYTES}-byte limit")
|
|
53
|
+
return models._safe_serialize(metadata)
|
|
43
54
|
|
|
44
55
|
|
|
45
56
|
def _raise_http_error(response: httpx.Response, action: str) -> None:
|
|
@@ -155,11 +166,11 @@ def deploy(module, metric=None, timeout=300, metadata=None):
|
|
|
155
166
|
# slot and reject an oversized deploy before signing it (Slice E).
|
|
156
167
|
"declared_content_length": len(zip_bytes),
|
|
157
168
|
}
|
|
158
|
-
# Carry the caller's metadata blob (validated
|
|
159
|
-
# reserved top-level body key, committed with the deployment
|
|
169
|
+
# Carry the caller's metadata blob (validated and made jsonb-storable at the
|
|
170
|
+
# call site) as a reserved top-level body key, committed with the deployment
|
|
171
|
+
# record server-side.
|
|
160
172
|
if metadata is not None:
|
|
161
|
-
|
|
162
|
-
deploy_body["metadata"] = metadata
|
|
173
|
+
deploy_body["metadata"] = _prepare_deploy_metadata(metadata)
|
|
163
174
|
if sig is not None:
|
|
164
175
|
deploy_body["signature_name"] = packaging._signature_to_string(sig)
|
|
165
176
|
deploy_body["module_class"] = type(module).__name__
|
|
@@ -62,21 +62,68 @@ def _send_json(client: httpx.Client, method: str, path: str, payload: Any) -> ht
|
|
|
62
62
|
# so a re-POST of a request that actually landed can't double-insert.
|
|
63
63
|
_RETRYABLE_STATUS = frozenset({502, 503, 504})
|
|
64
64
|
|
|
65
|
+
# Backpressure, not unavailability: the server's buffer is full and it wants the
|
|
66
|
+
# record dropped, so the caller counts it as *dropped* and we never retry it —
|
|
67
|
+
# not even if the body carried a retryable flag. Named because that exemption is
|
|
68
|
+
# the one thing the body-flag read below must not be able to override.
|
|
69
|
+
_DROP_STATUS = 429
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _server_says_retryable(response: httpx.Response) -> bool:
|
|
73
|
+
"""Whether the server explicitly marked this error response retryable.
|
|
74
|
+
|
|
75
|
+
The `"retryable": true` error-body field is the server's stable machine
|
|
76
|
+
signal for "this failed transiently and did not commit — send it again". It
|
|
77
|
+
is what keeps this side off a hard-coded status list: the server can start
|
|
78
|
+
signalling a new transient condition without an SDK release. A body that
|
|
79
|
+
isn't JSON, isn't an object, or lacks the field reads as not-retryable.
|
|
80
|
+
"""
|
|
81
|
+
try:
|
|
82
|
+
body = response.json()
|
|
83
|
+
except Exception:
|
|
84
|
+
return False
|
|
85
|
+
return isinstance(body, dict) and body.get("retryable") is True
|
|
86
|
+
|
|
65
87
|
|
|
66
88
|
def _is_retryable_export_error(exc: BaseException) -> bool:
|
|
67
89
|
"""Whether a failed data send is a transient delivery blip worth retrying.
|
|
68
90
|
|
|
69
|
-
True for gateway/overload statuses (502/503/504)
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
91
|
+
True for gateway/overload statuses (502/503/504), a response the server
|
|
92
|
+
explicitly flagged retryable (see `_server_says_retryable`), and httpx
|
|
93
|
+
transport/timeout errors (connect/read/pool timeouts, connection
|
|
94
|
+
refused/reset). False for everything else — 4xx (client error, a retry won't
|
|
95
|
+
help), 429 (backpressure, the caller records it as *dropped* not errored, and
|
|
96
|
+
no body flag overrides that), and an unflagged 500 (the platform's to answer
|
|
97
|
+
for, mirroring the is_external_outage boundary). Non-httpx exceptions are not
|
|
98
|
+
retried (a bug in our own build path shouldn't loop)."""
|
|
75
99
|
if isinstance(exc, httpx.HTTPStatusError):
|
|
76
|
-
|
|
100
|
+
status = exc.response.status_code
|
|
101
|
+
if status == _DROP_STATUS:
|
|
102
|
+
return False
|
|
103
|
+
return status in _RETRYABLE_STATUS or _server_says_retryable(exc.response)
|
|
77
104
|
return isinstance(exc, (httpx.TimeoutException, httpx.TransportError))
|
|
78
105
|
|
|
79
106
|
|
|
107
|
+
def _retry_after_seconds(exc: BaseException) -> float | None:
|
|
108
|
+
"""The server's own backoff hint for this failure, in seconds, or None.
|
|
109
|
+
|
|
110
|
+
Only the delta-seconds form of `Retry-After` is honored — that is what the
|
|
111
|
+
server emits alongside a retryable error. A missing header, the HTTP-date
|
|
112
|
+
form, or an unparseable value returns None so the caller falls back to its
|
|
113
|
+
own exponential backoff rather than trusting a number it can't read. A
|
|
114
|
+
transport error has no response at all and so has no hint.
|
|
115
|
+
"""
|
|
116
|
+
if not isinstance(exc, httpx.HTTPStatusError):
|
|
117
|
+
return None
|
|
118
|
+
raw = exc.response.headers.get("retry-after")
|
|
119
|
+
if not raw:
|
|
120
|
+
return None
|
|
121
|
+
try:
|
|
122
|
+
return max(0.0, float(raw))
|
|
123
|
+
except ValueError:
|
|
124
|
+
return None
|
|
125
|
+
|
|
126
|
+
|
|
80
127
|
@dataclass
|
|
81
128
|
class ExportItem:
|
|
82
129
|
"""Item in the export queue."""
|
|
@@ -144,6 +191,12 @@ class BatchExporter:
|
|
|
144
191
|
# can zero the backoff. Distinct from the start-post budget above.
|
|
145
192
|
self._data_post_attempts = 3
|
|
146
193
|
self._data_post_backoff = 0.5
|
|
194
|
+
# Ceiling on a single inter-attempt wait, including one the server asked
|
|
195
|
+
# for via Retry-After. The exporter thread is the only thing draining the
|
|
196
|
+
# queue, so an over-long wait doesn't just delay this send — it stalls
|
|
197
|
+
# every queued record behind it and risks the queue filling. Honoring the
|
|
198
|
+
# server's hint is worth it; honoring an unbounded one is not.
|
|
199
|
+
self._data_post_max_delay = 5.0
|
|
147
200
|
|
|
148
201
|
# Guard against registering multiple exit handlers on repeated start()
|
|
149
202
|
self._atexit_registered = False
|
|
@@ -473,13 +526,16 @@ class BatchExporter:
|
|
|
473
526
|
def _send_json_retrying(self, client: httpx.Client, method: str, path: str, payload: Any) -> httpx.Response:
|
|
474
527
|
"""`_send_json` + `raise_for_status`, retrying transient delivery blips.
|
|
475
528
|
|
|
476
|
-
Retries `_data_post_attempts` times
|
|
477
|
-
|
|
478
|
-
`_is_retryable_export_error`)
|
|
479
|
-
|
|
480
|
-
|
|
481
|
-
|
|
482
|
-
|
|
529
|
+
Retries `_data_post_attempts` times on a retryable failure — a
|
|
530
|
+
gateway/overload status, a response the server flagged `retryable`, or a
|
|
531
|
+
transport/timeout error (see `_is_retryable_export_error`) — then
|
|
532
|
+
re-raises. The wait between attempts is the server's `Retry-After` hint
|
|
533
|
+
when it sent one, else exponential backoff, capped either way by
|
|
534
|
+
`_data_post_max_delay`. A non-retryable failure (4xx, 429, an unflagged
|
|
535
|
+
500, or a 2xx that somehow raised) propagates on the first attempt, so
|
|
536
|
+
the caller's existing except blocks still record it (429 → dropped,
|
|
537
|
+
everything else → error) exactly as before — the only change is that a
|
|
538
|
+
transient blip now gets a few tries before it counts."""
|
|
483
539
|
last_exc: Exception | None = None
|
|
484
540
|
for attempt in range(self._data_post_attempts):
|
|
485
541
|
try:
|
|
@@ -489,10 +545,19 @@ class BatchExporter:
|
|
|
489
545
|
except Exception as e:
|
|
490
546
|
last_exc = e
|
|
491
547
|
if _is_retryable_export_error(e) and attempt + 1 < self._data_post_attempts:
|
|
548
|
+
delay = _retry_after_seconds(e)
|
|
549
|
+
if delay is None:
|
|
550
|
+
delay = self._data_post_backoff * (2**attempt)
|
|
551
|
+
delay = min(delay, self._data_post_max_delay)
|
|
492
552
|
logger.debug(
|
|
493
|
-
"cmpnd: retrying %s %s after transient error (attempt %d): %s",
|
|
553
|
+
"cmpnd: retrying %s %s after transient error in %.2fs (attempt %d): %s",
|
|
554
|
+
method,
|
|
555
|
+
path,
|
|
556
|
+
delay,
|
|
557
|
+
attempt + 1,
|
|
558
|
+
e,
|
|
494
559
|
)
|
|
495
|
-
time.sleep(
|
|
560
|
+
time.sleep(delay)
|
|
496
561
|
continue
|
|
497
562
|
raise
|
|
498
563
|
assert last_exc is not None # unreachable: the loop either returns or raises
|
|
@@ -3,6 +3,7 @@
|
|
|
3
3
|
from __future__ import annotations
|
|
4
4
|
|
|
5
5
|
import json
|
|
6
|
+
import re
|
|
6
7
|
from dataclasses import dataclass, field
|
|
7
8
|
from datetime import datetime, timezone
|
|
8
9
|
from enum import Enum
|
|
@@ -11,6 +12,50 @@ from uuid import UUID
|
|
|
11
12
|
|
|
12
13
|
from . import identity, imaging
|
|
13
14
|
|
|
15
|
+
# U+0000 (NUL) and lone UTF-16 surrogates are valid JSON, and valid in a Python
|
|
16
|
+
# str, but the server cannot store either: a JSON column rejects both, and a NUL
|
|
17
|
+
# is rejected by a plain text column too (it is valid UTF-8, but PostgreSQL text
|
|
18
|
+
# cannot hold it). A captured input/output/attribute string, and the text of an
|
|
19
|
+
# exception quoting that data, can all legitimately contain them — so we strip
|
|
20
|
+
# them here rather than shipping a payload the server must reject. On a UTF-8
|
|
21
|
+
# database these are the only such code points, so this is the complete set.
|
|
22
|
+
#
|
|
23
|
+
# "Chokepoint" is load-bearing and is a property of the callers, not just of
|
|
24
|
+
# this function: every payload field bound for a JSON *or* text column goes
|
|
25
|
+
# through _safe_serialize (or _strip_unstorable / _strip_unstorable_opt for a
|
|
26
|
+
# value already known to be a string). Anything that hand-rolls its own
|
|
27
|
+
# serializer instead reopens the hole — the optimizer tracker and the
|
|
28
|
+
# dataset/deploy payload builders therefore route through here too.
|
|
29
|
+
_UNSTORABLE_CODEPOINTS = re.compile("[\x00\ud800-\udfff]")
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _strip_unstorable(s: str) -> str:
|
|
33
|
+
"""Drop un-storable code points (U+0000, lone surrogates) from ``s``.
|
|
34
|
+
|
|
35
|
+
Fast path: the C-level search returns ``None`` for the ~always-clean case,
|
|
36
|
+
so no replacement string is built.
|
|
37
|
+
"""
|
|
38
|
+
return _UNSTORABLE_CODEPOINTS.sub("", s) if _UNSTORABLE_CODEPOINTS.search(s) else s
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _strip_unstorable_opt(s: str | None) -> str | None:
|
|
42
|
+
"""``_strip_unstorable`` for an optional string, preserving ``None``.
|
|
43
|
+
|
|
44
|
+
For the top-level payload fields that are plain strings on the wire and land
|
|
45
|
+
in text columns (notably ``error_message``, which is ``str(exception)`` and
|
|
46
|
+
so routinely quotes the very captured data the code point came from) — they
|
|
47
|
+
never pass through ``_safe_serialize``.
|
|
48
|
+
"""
|
|
49
|
+
return None if s is None else _strip_unstorable(s)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _type_name(value: Any) -> str:
|
|
53
|
+
"""Scrubbed class name of ``value``, for the type tags and marker strings
|
|
54
|
+
below. CPython refuses an un-storable code point in a real class name, so
|
|
55
|
+
this is defensive: a metaclass answering ``__name__`` itself is the only way
|
|
56
|
+
one arrives, and the scrub costs one call at each interpolation site."""
|
|
57
|
+
return _strip_unstorable(type(value).__name__)
|
|
58
|
+
|
|
14
59
|
|
|
15
60
|
def _safe_serialize(value: Any, _stack: set[int] | None = None) -> Any:
|
|
16
61
|
"""Recursively serialize a value to JSON-safe types.
|
|
@@ -28,8 +73,11 @@ def _safe_serialize(value: Any, _stack: set[int] | None = None) -> Any:
|
|
|
28
73
|
if value is None:
|
|
29
74
|
return None
|
|
30
75
|
|
|
31
|
-
# Primitives - JSON safe (no need to track these)
|
|
32
|
-
|
|
76
|
+
# Primitives - JSON safe (no need to track these). Strings are scrubbed of
|
|
77
|
+
# jsonb-un-storable code points; numerics/bools can't carry any.
|
|
78
|
+
if isinstance(value, str):
|
|
79
|
+
return _strip_unstorable(value)
|
|
80
|
+
if isinstance(value, (int, float, bool)):
|
|
33
81
|
return value
|
|
34
82
|
|
|
35
83
|
# UUID
|
|
@@ -43,23 +91,27 @@ def _safe_serialize(value: Any, _stack: set[int] | None = None) -> Any:
|
|
|
43
91
|
# Bytes
|
|
44
92
|
if isinstance(value, bytes):
|
|
45
93
|
try:
|
|
46
|
-
return value.decode("utf-8")
|
|
94
|
+
return _strip_unstorable(value.decode("utf-8"))
|
|
47
95
|
except Exception:
|
|
48
96
|
return f"<bytes: {len(value)} bytes>"
|
|
49
97
|
|
|
50
|
-
# Enums
|
|
98
|
+
# Enums — recurse so a string-backed member value is scrubbed (real enums
|
|
99
|
+
# like SpanType have clean values, so this is a no-op for them; it only
|
|
100
|
+
# strips a pathological NUL/lone-surrogate value).
|
|
51
101
|
if isinstance(value, Enum):
|
|
52
|
-
return value.value
|
|
102
|
+
return _safe_serialize(value.value, _stack)
|
|
53
103
|
|
|
54
|
-
# Types/classes (like SignatureMeta)
|
|
104
|
+
# Types/classes (like SignatureMeta). repr() of a class is whatever its
|
|
105
|
+
# metaclass says — DSPy's interpolates the signature's instructions and field
|
|
106
|
+
# metadata — so it is captured text, not a bounded label.
|
|
55
107
|
if isinstance(value, type):
|
|
56
|
-
return repr(value)
|
|
108
|
+
return _strip_unstorable(repr(value))
|
|
57
109
|
|
|
58
110
|
# For compound types, check for cycles using object identity
|
|
59
111
|
obj_id = id(value)
|
|
60
112
|
if obj_id in _stack:
|
|
61
113
|
# True cycle detected - we're currently serializing this object
|
|
62
|
-
return f"<circular ref: {
|
|
114
|
+
return f"<circular ref: {_type_name(value)}>"
|
|
63
115
|
|
|
64
116
|
# Add to stack before recursing into compound types
|
|
65
117
|
_stack.add(obj_id)
|
|
@@ -73,9 +125,9 @@ def _safe_serialize(value: Any, _stack: set[int] | None = None) -> Any:
|
|
|
73
125
|
if isinstance(value, (list, tuple)):
|
|
74
126
|
return [_safe_serialize(v, _stack) for v in value]
|
|
75
127
|
|
|
76
|
-
# Dicts
|
|
128
|
+
# Dicts (keys become jsonb object keys, so scrub them too)
|
|
77
129
|
if isinstance(value, dict):
|
|
78
|
-
return {str(k): _safe_serialize(v, _stack) for k, v in value.items()}
|
|
130
|
+
return {_strip_unstorable(str(k)): _safe_serialize(v, _stack) for k, v in value.items()}
|
|
79
131
|
|
|
80
132
|
# DSPy Prediction and similar - has toDict method
|
|
81
133
|
if hasattr(value, "toDict") and callable(value.toDict):
|
|
@@ -107,12 +159,17 @@ def _safe_serialize(value: Any, _stack: set[int] | None = None) -> Any:
|
|
|
107
159
|
except Exception:
|
|
108
160
|
pass
|
|
109
161
|
|
|
110
|
-
# Objects with __dict__ (general objects, custom classes)
|
|
162
|
+
# Objects with __dict__ (general objects, custom classes). Attribute
|
|
163
|
+
# names become jsonb object keys, so scrub them like dict keys.
|
|
111
164
|
if hasattr(value, "__dict__"):
|
|
112
165
|
try:
|
|
113
|
-
obj_dict = {
|
|
166
|
+
obj_dict = {
|
|
167
|
+
_strip_unstorable(k): _safe_serialize(v, _stack)
|
|
168
|
+
for k, v in value.__dict__.items()
|
|
169
|
+
if not k.startswith("_")
|
|
170
|
+
}
|
|
114
171
|
# Add type info for clarity
|
|
115
|
-
obj_dict["__type__"] =
|
|
172
|
+
obj_dict["__type__"] = _type_name(value)
|
|
116
173
|
return obj_dict
|
|
117
174
|
except Exception:
|
|
118
175
|
pass
|
|
@@ -123,17 +180,17 @@ def _safe_serialize(value: Any, _stack: set[int] | None = None) -> Any:
|
|
|
123
180
|
obj_dict = {}
|
|
124
181
|
for slot in value.__slots__: # pyright: ignore[reportAttributeAccessIssue]
|
|
125
182
|
if not slot.startswith("_") and hasattr(value, slot):
|
|
126
|
-
obj_dict[slot] = _safe_serialize(getattr(value, slot), _stack)
|
|
127
|
-
obj_dict["__type__"] =
|
|
183
|
+
obj_dict[_strip_unstorable(slot)] = _safe_serialize(getattr(value, slot), _stack)
|
|
184
|
+
obj_dict["__type__"] = _type_name(value)
|
|
128
185
|
return obj_dict
|
|
129
186
|
except Exception:
|
|
130
187
|
pass
|
|
131
188
|
|
|
132
189
|
# Final fallback - convert to string
|
|
133
190
|
try:
|
|
134
|
-
return str(value)
|
|
191
|
+
return _strip_unstorable(str(value))
|
|
135
192
|
except Exception:
|
|
136
|
-
return f"<unserializable: {
|
|
193
|
+
return f"<unserializable: {_type_name(value)}>"
|
|
137
194
|
|
|
138
195
|
finally:
|
|
139
196
|
# Remove from stack after we're done serializing this object
|
|
@@ -455,7 +512,13 @@ class SpanBuilder:
|
|
|
455
512
|
)
|
|
456
513
|
|
|
457
514
|
def _build_attributes(self) -> dict[str, str]:
|
|
458
|
-
"""Build attributes dict, injecting repl_history if present.
|
|
515
|
+
"""Build attributes dict, injecting repl_history if present.
|
|
516
|
+
|
|
517
|
+
The nested JSON rides as an attribute *value*, so it reaches the wire
|
|
518
|
+
through the same scrub as every other attribute string (``to_api_dict``
|
|
519
|
+
passes the result to ``_safe_serialize``) — the jsonb-storability of the
|
|
520
|
+
column does not rest on ``json.dumps``' ``ensure_ascii`` escaping here.
|
|
521
|
+
"""
|
|
459
522
|
attrs = dict(self.attributes)
|
|
460
523
|
if self.repl_history:
|
|
461
524
|
attrs["repl_history"] = json.dumps(self.repl_history)
|
|
@@ -472,7 +535,7 @@ class SpanBuilder:
|
|
|
472
535
|
"start_time": self.start_time.isoformat() if self.start_time else None,
|
|
473
536
|
"end_time": self.end_time.isoformat() if self.end_time else None,
|
|
474
537
|
"status": self.status.value,
|
|
475
|
-
"error_message": self.error_message,
|
|
538
|
+
"error_message": _strip_unstorable_opt(self.error_message),
|
|
476
539
|
"error_type": self.error_type,
|
|
477
540
|
"signature_name": self.signature_name,
|
|
478
541
|
"signature_instructions": self.signature_instructions,
|
|
@@ -577,7 +640,7 @@ class TraceBuilder:
|
|
|
577
640
|
"start_time": self.start_time.isoformat() if self.start_time else None,
|
|
578
641
|
"end_time": self.end_time.isoformat() if self.end_time else None,
|
|
579
642
|
"status": self.status.value,
|
|
580
|
-
"error_message": self.error_message,
|
|
643
|
+
"error_message": _strip_unstorable_opt(self.error_message),
|
|
581
644
|
"program_name": self.program_name,
|
|
582
645
|
"program_version": self.program_version,
|
|
583
646
|
"signature_name": self.signature_name,
|
|
@@ -589,7 +652,11 @@ class TraceBuilder:
|
|
|
589
652
|
"signature_output_field_schemas": self.signature_output_field_schemas,
|
|
590
653
|
"program_signature_hash": identity.format_signature_hash(self.program_signature_hash),
|
|
591
654
|
"signature_summary": self.signature_summary,
|
|
592
|
-
|
|
655
|
+
# The server forwards this blob verbatim into a jsonb column, and its
|
|
656
|
+
# program registration is best-effort (a rejection is only logged), so
|
|
657
|
+
# an un-storable code point here would silently lose the whole
|
|
658
|
+
# component list. Scrub at the source — the only place that can.
|
|
659
|
+
"component_signatures": _safe_serialize(self.component_signatures),
|
|
593
660
|
"compiled_from_run_id": self.compiled_from_run_id,
|
|
594
661
|
"demo_ids": self.demo_ids,
|
|
595
662
|
"request_preview": (self.request_preview or "")[:500],
|
|
@@ -597,8 +664,8 @@ class TraceBuilder:
|
|
|
597
664
|
"eval_run_id": str(self.eval_run_id) if self.eval_run_id else None,
|
|
598
665
|
"optimization_run_id": (str(self.optimization_run_id) if self.optimization_run_id else None),
|
|
599
666
|
"adapter_type": self.adapter_type,
|
|
600
|
-
"tags": self.tags,
|
|
601
|
-
"metadata": self.metadata,
|
|
667
|
+
"tags": _safe_serialize(self.tags),
|
|
668
|
+
"metadata": _safe_serialize(self.metadata),
|
|
602
669
|
"project": self.project_tags[0] if self.project_tags else None,
|
|
603
670
|
"project_tags": self.project_tags if self.project_tags else [],
|
|
604
671
|
"spans": [s.to_api_dict() for s in self._batch_spans] if self._batch_spans else [],
|
|
@@ -617,7 +684,7 @@ class TraceBuilder:
|
|
|
617
684
|
"signature_output_field_types": self.signature_output_field_types,
|
|
618
685
|
"program_signature_hash": identity.format_signature_hash(self.program_signature_hash),
|
|
619
686
|
"signature_summary": self.signature_summary,
|
|
620
|
-
"component_signatures": self.component_signatures,
|
|
687
|
+
"component_signatures": _safe_serialize(self.component_signatures),
|
|
621
688
|
"root_span_type": self.root_span_type,
|
|
622
689
|
"project": self.project_tags[0] if self.project_tags else None,
|
|
623
690
|
"project_tags": self.project_tags if self.project_tags else [],
|
|
@@ -629,7 +696,7 @@ class TraceBuilder:
|
|
|
629
696
|
"start_time": self.start_time.isoformat() if self.start_time else None,
|
|
630
697
|
"end_time": self.end_time.isoformat() if self.end_time else None,
|
|
631
698
|
"status": self.status.value,
|
|
632
|
-
"error_message": self.error_message,
|
|
699
|
+
"error_message": _strip_unstorable_opt(self.error_message),
|
|
633
700
|
"request_preview": (self.request_preview or "")[:500],
|
|
634
701
|
"response_preview": (self.response_preview or "")[:500],
|
|
635
702
|
"total_tokens": self.total_tokens,
|
|
@@ -642,8 +709,8 @@ class TraceBuilder:
|
|
|
642
709
|
"demo_ids": self.demo_ids,
|
|
643
710
|
"eval_run_id": str(self.eval_run_id) if self.eval_run_id else None,
|
|
644
711
|
"optimization_run_id": (str(self.optimization_run_id) if self.optimization_run_id else None),
|
|
645
|
-
"tags": self.tags,
|
|
646
|
-
"metadata": self.metadata,
|
|
712
|
+
"tags": _safe_serialize(self.tags),
|
|
713
|
+
"metadata": _safe_serialize(self.metadata),
|
|
647
714
|
"project": self.project_tags[0] if self.project_tags else None,
|
|
648
715
|
"project_tags": self.project_tags if self.project_tags else [],
|
|
649
716
|
}
|
|
@@ -757,7 +824,7 @@ class EvalExampleBuilder:
|
|
|
757
824
|
"total_tokens": self.total_tokens,
|
|
758
825
|
"cost": self.cost,
|
|
759
826
|
"status": self.status.value,
|
|
760
|
-
"error_message": self.error_message,
|
|
827
|
+
"error_message": _strip_unstorable_opt(self.error_message),
|
|
761
828
|
}
|
|
762
829
|
|
|
763
830
|
|
|
@@ -922,7 +989,7 @@ class EvalRunBuilder:
|
|
|
922
989
|
"end_time": self.end_time.isoformat() if self.end_time else None,
|
|
923
990
|
"duration_ms": self.duration_ms,
|
|
924
991
|
"status": self.status.value,
|
|
925
|
-
"error_message": self.error_message,
|
|
992
|
+
"error_message": _strip_unstorable_opt(self.error_message),
|
|
926
993
|
"score": self.score,
|
|
927
994
|
"total_examples": self.total_examples,
|
|
928
995
|
"passed_examples": self.passed_examples,
|
|
@@ -935,8 +1002,8 @@ class EvalRunBuilder:
|
|
|
935
1002
|
"total_cost": self.total_cost,
|
|
936
1003
|
"avg_tokens_per_example": self.avg_tokens_per_example,
|
|
937
1004
|
"avg_cost_per_example": self.avg_cost_per_example,
|
|
938
|
-
"tags": self.tags,
|
|
939
|
-
"metadata": self.metadata,
|
|
1005
|
+
"tags": _safe_serialize(self.tags),
|
|
1006
|
+
"metadata": _safe_serialize(self.metadata),
|
|
940
1007
|
"examples": [ex.to_api_dict() for ex in self._examples],
|
|
941
1008
|
"optimization_run_id": (str(self.optimization_run_id) if self.optimization_run_id else None),
|
|
942
1009
|
"project": self.project_tags[0] if self.project_tags else None,
|
|
@@ -12,13 +12,12 @@ Usage:
|
|
|
12
12
|
|
|
13
13
|
from __future__ import annotations
|
|
14
14
|
|
|
15
|
-
import json
|
|
16
15
|
import logging
|
|
17
16
|
import time
|
|
18
17
|
from typing import Any
|
|
19
18
|
from uuid import UUID
|
|
20
19
|
|
|
21
|
-
from .. import context
|
|
20
|
+
from .. import context, models
|
|
22
21
|
from . import tracker
|
|
23
22
|
|
|
24
23
|
logger = logging.getLogger(__name__)
|
|
@@ -191,8 +190,10 @@ class CmpndGEPACallback:
|
|
|
191
190
|
components = event.get("components", [])
|
|
192
191
|
dataset = event.get("dataset", {})
|
|
193
192
|
|
|
194
|
-
# Deep copy to avoid mutation (matches tracker.py pattern)
|
|
195
|
-
|
|
193
|
+
# Deep copy to avoid mutation (matches tracker.py pattern), and the
|
|
194
|
+
# copy is jsonb-storable: the dataset is raw captured LM text bound
|
|
195
|
+
# for a jsonb column.
|
|
196
|
+
dataset_copy = models._safe_serialize(dataset)
|
|
196
197
|
|
|
197
198
|
self._tracker.capture_reflective_dataset(
|
|
198
199
|
iteration=iteration,
|
|
@@ -464,17 +465,9 @@ class CmpndGEPACallback:
|
|
|
464
465
|
|
|
465
466
|
for idx, instructions in enumerate(candidates):
|
|
466
467
|
try:
|
|
467
|
-
# Extract parent
|
|
468
|
-
|
|
469
|
-
if idx < len(parents)
|
|
470
|
-
parent_list = parents[idx]
|
|
471
|
-
if isinstance(parent_list, (list, tuple)):
|
|
472
|
-
for p in parent_list:
|
|
473
|
-
if p is not None:
|
|
474
|
-
parent_idx = p
|
|
475
|
-
break
|
|
476
|
-
elif parent_list is not None:
|
|
477
|
-
parent_idx = parent_list
|
|
468
|
+
# Extract the full parent list — a merge proposal has two, and
|
|
469
|
+
# serve derives the mutation type and generation from them.
|
|
470
|
+
parents_of = tracker.normalize_parent_indices(parents[idx] if idx < len(parents) else None)
|
|
478
471
|
|
|
479
472
|
# Compute aggregate score: prefer val_subscores (per-example),
|
|
480
473
|
# fall back to objective_scores (per-objective)
|
|
@@ -503,7 +496,7 @@ class CmpndGEPACallback:
|
|
|
503
496
|
tracker.CandidateData(
|
|
504
497
|
candidate_idx=idx,
|
|
505
498
|
instructions=instructions if isinstance(instructions, dict) else {},
|
|
506
|
-
|
|
499
|
+
parent_indices=parents_of,
|
|
507
500
|
discovery_iteration=int(discovery_counts[idx]) if idx < len(discovery_counts) else 0,
|
|
508
501
|
val_aggregate_score=val_aggregate_score,
|
|
509
502
|
val_subscores=val_subs,
|