agentevolve-optimizer 0.5.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_evolve/__init__.py +722 -0
- agent_evolve/agentic.py +2800 -0
- agent_evolve/api.py +767 -0
- agent_evolve/application/__init__.py +1580 -0
- agent_evolve/application/action_allocation.py +744 -0
- agent_evolve/application/action_allocation_frame.py +347 -0
- agent_evolve/application/action_allocation_frame_commit.py +185 -0
- agent_evolve/application/action_allocation_frame_commit_v3.py +184 -0
- agent_evolve/application/action_allocation_frame_v3.py +338 -0
- agent_evolve/application/action_archive_value.py +497 -0
- agent_evolve/application/action_evidence_consistency.py +455 -0
- agent_evolve/application/action_forecast_partitioning.py +1471 -0
- agent_evolve/application/action_metric_projection.py +211 -0
- agent_evolve/application/action_role_value.py +680 -0
- agent_evolve/application/action_score_authorities.py +363 -0
- agent_evolve/application/action_structural_signature.py +116 -0
- agent_evolve/application/action_target_realization.py +402 -0
- agent_evolve/application/agentic_evolution.py +7734 -0
- agent_evolve/application/agentic_portfolio_residual_expert.py +835 -0
- agent_evolve/application/anchor_residual_identification.py +463 -0
- agent_evolve/application/archive_conditioned_action_target.py +208 -0
- agent_evolve/application/artifact_journal.py +246 -0
- agent_evolve/application/artifact_replay.py +347 -0
- agent_evolve/application/budgeted_optimizer.py +1828 -0
- agent_evolve/application/calibrated_campaign.py +485 -0
- agent_evolve/application/calibrated_current_prefix_forecast_opportunity.py +322 -0
- agent_evolve/application/calibrated_positive_gain_opportunity.py +1581 -0
- agent_evolve/application/campaign_capacity_recourse.py +254 -0
- agent_evolve/application/campaign_contextual_outcomes.py +119 -0
- agent_evolve/application/campaign_diagnostic_blocks.py +930 -0
- agent_evolve/application/campaign_evidence_registry.py +262 -0
- agent_evolve/application/campaign_execution.py +2537 -0
- agent_evolve/application/campaign_generation_audit.py +942 -0
- agent_evolve/application/campaign_learning.py +1812 -0
- agent_evolve/application/campaign_learning_runtime.py +1977 -0
- agent_evolve/application/campaign_search_phase.py +227 -0
- agent_evolve/application/campaign_selector_context_extension.py +220 -0
- agent_evolve/application/campaign_variation_envelope.py +649 -0
- agent_evolve/application/campaign_variation_trace.py +451 -0
- agent_evolve/application/candidate_archive_consequence.py +128 -0
- agent_evolve/application/causal_opportunity_portfolio_gate.py +385 -0
- agent_evolve/application/composite_outcome_updater.py +145 -0
- agent_evolve/application/composition_portfolio_selection.py +363 -0
- agent_evolve/application/concurrent_stage.py +144 -0
- agent_evolve/application/contextual_action_allocation.py +181 -0
- agent_evolve/application/contextual_campaign_outcomes.py +267 -0
- agent_evolve/application/contextual_campaign_planning.py +1366 -0
- agent_evolve/application/contextual_delayed_credit.py +651 -0
- agent_evolve/application/contextual_search_controller.py +2374 -0
- agent_evolve/application/current_prefix_forecast_opportunity.py +714 -0
- agent_evolve/application/decision_metric_projection.py +112 -0
- agent_evolve/application/derived_action_semantics.py +129 -0
- agent_evolve/application/detailed_evaluation.py +449 -0
- agent_evolve/application/earned_lineage.py +1011 -0
- agent_evolve/application/effective_choice_audit.py +484 -0
- agent_evolve/application/empirical_consequence_calibration.py +908 -0
- agent_evolve/application/evaluation_accounting.py +325 -0
- agent_evolve/application/evaluation_cache.py +199 -0
- agent_evolve/application/evaluation_escrow.py +547 -0
- agent_evolve/application/evaluation_recourse.py +253 -0
- agent_evolve/application/event_recorder.py +151 -0
- agent_evolve/application/evolution_campaign.py +1840 -0
- agent_evolve/application/executable_hypothesis.py +323 -0
- agent_evolve/application/factorial_branch_pilot.py +772 -0
- agent_evolve/application/finite_acquisition_capacity_recourse.py +672 -0
- agent_evolve/application/finite_acquisition_residual_expert.py +373 -0
- agent_evolve/application/finite_acquisition_variation_envelope.py +802 -0
- agent_evolve/application/finite_action_hypothesis_semantics.py +446 -0
- agent_evolve/application/finite_action_selection.py +188 -0
- agent_evolve/application/finite_action_set.py +306 -0
- agent_evolve/application/finite_action_transition.py +537 -0
- agent_evolve/application/finite_variation_eligibility.py +296 -0
- agent_evolve/application/forecast_geometry_portfolio.py +799 -0
- agent_evolve/application/forecast_opportunity_shadow_calibration.py +316 -0
- agent_evolve/application/front_proximity_admission.py +311 -0
- agent_evolve/application/front_proximity_parent_basis.py +458 -0
- agent_evolve/application/frozen_hurdle_score.py +659 -0
- agent_evolve/application/g3_causal_screen.py +2257 -0
- agent_evolve/application/g3_causal_validation.py +1046 -0
- agent_evolve/application/g3_postseal_curation.py +818 -0
- agent_evolve/application/gated_agentic_generator.py +205 -0
- agent_evolve/application/generation_feedback.py +293 -0
- agent_evolve/application/generative_proposal_journal.py +185 -0
- agent_evolve/application/geometry_conditional_elasticity.py +453 -0
- agent_evolve/application/global_wave_action_allocation.py +1151 -0
- agent_evolve/application/head_mass_conditional_seat.py +268 -0
- agent_evolve/application/identifiable_reflection_evidence.py +1147 -0
- agent_evolve/application/identifiable_reflection_learning.py +395 -0
- agent_evolve/application/identifiable_reflection_request.py +364 -0
- agent_evolve/application/in_memory_residual_archive.py +341 -0
- agent_evolve/application/insight_memory.py +1804 -0
- agent_evolve/application/live_runtime_manifest.py +758 -0
- agent_evolve/application/llm_task_queue.py +769 -0
- agent_evolve/application/matched_finite_action_block.py +409 -0
- agent_evolve/application/materialized_action_broker.py +2328 -0
- agent_evolve/application/materialized_action_constraints.py +83 -0
- agent_evolve/application/materialized_variation.py +211 -0
- agent_evolve/application/multi_option_evolution.py +1536 -0
- agent_evolve/application/outcome_adaptive_action_racing.py +2827 -0
- agent_evolve/application/outcome_adaptive_residual_campaign_runtime.py +580 -0
- agent_evolve/application/outcome_adaptive_residual_portfolio_evolution.py +3671 -0
- agent_evolve/application/outcome_conditioned_portfolio_selection.py +1374 -0
- agent_evolve/application/outcome_relation.py +193 -0
- agent_evolve/application/paired_allocation_comparison.py +241 -0
- agent_evolve/application/paired_block_schedule.py +127 -0
- agent_evolve/application/parent_measurement.py +226 -0
- agent_evolve/application/pareto_archive.py +811 -0
- agent_evolve/application/portfolio_campaign_runtime.py +4739 -0
- agent_evolve/application/portfolio_evolution.py +2950 -0
- agent_evolve/application/portfolio_hypothesis_observations.py +814 -0
- agent_evolve/application/portfolio_memory_attribution.py +581 -0
- agent_evolve/application/portfolio_memory_dose.py +788 -0
- agent_evolve/application/portfolio_memory_matched_control.py +938 -0
- agent_evolve/application/portfolio_memory_transfer.py +297 -0
- agent_evolve/application/portfolio_optimization_memory.py +363 -0
- agent_evolve/application/portfolio_outcome_feedback.py +1613 -0
- agent_evolve/application/portfolio_projection.py +335 -0
- agent_evolve/application/portfolio_recombination.py +2032 -0
- agent_evolve/application/post_evolution_reflection.py +834 -0
- agent_evolve/application/postcommit_rank_authority.py +245 -0
- agent_evolve/application/precommitted_portfolio_racing.py +2762 -0
- agent_evolve/application/prequential_archive_opportunity_calibration.py +1154 -0
- agent_evolve/application/prequential_residual_exploration.py +343 -0
- agent_evolve/application/prequential_score_portfolio.py +954 -0
- agent_evolve/application/projections.py +292 -0
- agent_evolve/application/protected_action_committee.py +1027 -0
- agent_evolve/application/protected_branch_pilot.py +376 -0
- agent_evolve/application/protected_current_prefix_forecast_opportunity.py +552 -0
- agent_evolve/application/provider_replay.py +910 -0
- agent_evolve/application/rank_balanced_causal_pilot.py +1372 -0
- agent_evolve/application/recombination_residual_expert.py +403 -0
- agent_evolve/application/reflection_workflow.py +571 -0
- agent_evolve/application/region_conditional_credit.py +911 -0
- agent_evolve/application/residual_campaign_runtime.py +531 -0
- agent_evolve/application/residual_headroom_campaign_runtime.py +459 -0
- agent_evolve/application/residual_headroom_ledger.py +1544 -0
- agent_evolve/application/residual_learning_transaction.py +396 -0
- agent_evolve/application/residual_portfolio_evolution.py +1228 -0
- agent_evolve/application/residual_reachability.py +749 -0
- agent_evolve/application/residual_stage_credit.py +499 -0
- agent_evolve/application/same_prefix_paired_audit.py +1580 -0
- agent_evolve/application/semantic_coverage_score_portfolio.py +838 -0
- agent_evolve/application/sequential_lineage_allocation.py +1017 -0
- agent_evolve/application/sequential_market_replay.py +1395 -0
- agent_evolve/application/sequential_residual_campaign_runtime.py +305 -0
- agent_evolve/application/sequential_residual_portfolio_evolution.py +940 -0
- agent_evolve/application/single_score_action_allocation.py +299 -0
- agent_evolve/application/source_exposure_allocation.py +906 -0
- agent_evolve/application/staged_memory.py +210 -0
- agent_evolve/application/stratified_cold_start_allocation.py +732 -0
- agent_evolve/application/support_guarded_hurdle_score.py +549 -0
- agent_evolve/application/target_conditioned_action_forecast.py +595 -0
- agent_evolve/application/target_conditioned_campaign.py +566 -0
- agent_evolve/application/treatment_assignment.py +201 -0
- agent_evolve/application/trusted_objective_evidence.py +217 -0
- agent_evolve/application/two_stage_action_evolution.py +1131 -0
- agent_evolve/application/v8lite_allocation_policy.py +1083 -0
- agent_evolve/application/v9_candidate_policy.py +1303 -0
- agent_evolve/bootstrap.py +108 -0
- agent_evolve/campaign_presets.py +517 -0
- agent_evolve/campaign_profiles.py +452 -0
- agent_evolve/campaign_variation_topology.py +288 -0
- agent_evolve/campaign_workload.py +950 -0
- agent_evolve/cli.py +797 -0
- agent_evolve/contract.py +241 -0
- agent_evolve/core/__init__.py +91 -0
- agent_evolve/core/action_semantics.py +411 -0
- agent_evolve/core/authored.py +105 -0
- agent_evolve/core/formatting.py +286 -0
- agent_evolve/core/optimization_semantics.py +324 -0
- agent_evolve/core/problem.py +167 -0
- agent_evolve/core/results.py +323 -0
- agent_evolve/core/stats.py +70 -0
- agent_evolve/core/telemetry.py +100 -0
- agent_evolve/domain/__init__.py +89 -0
- agent_evolve/domain/artifact.py +162 -0
- agent_evolve/domain/durable_text.py +68 -0
- agent_evolve/domain/event.py +1454 -0
- agent_evolve/domain/finite_action_set.py +426 -0
- agent_evolve/domain/finite_variation.py +526 -0
- agent_evolve/domain/generative_emission.py +559 -0
- agent_evolve/domain/ids.py +163 -0
- agent_evolve/domain/inline_text.py +106 -0
- agent_evolve/domain/insight.py +27 -0
- agent_evolve/domain/lineage.py +737 -0
- agent_evolve/domain/llm_task_queue.py +960 -0
- agent_evolve/domain/outcome.py +96 -0
- agent_evolve/domain/patch.py +854 -0
- agent_evolve/domain/typed_json.py +542 -0
- agent_evolve/domain/variation_space.py +158 -0
- agent_evolve/driver.py +1014 -0
- agent_evolve/harness/__init__.py +29 -0
- agent_evolve/harness/base.py +242 -0
- agent_evolve/harness/directives.py +163 -0
- agent_evolve/harness/generative_seal.py +479 -0
- agent_evolve/harness/registry.py +41 -0
- agent_evolve/infrastructure/__init__.py +39 -0
- agent_evolve/infrastructure/artifacts/__init__.py +6 -0
- agent_evolve/infrastructure/artifacts/_verification.py +67 -0
- agent_evolve/infrastructure/artifacts/filesystem.py +343 -0
- agent_evolve/infrastructure/artifacts/in_memory.py +73 -0
- agent_evolve/infrastructure/asyncio_runtime.py +109 -0
- agent_evolve/infrastructure/authored_runtime.py +188 -0
- agent_evolve/infrastructure/authored_worker.py +171 -0
- agent_evolve/infrastructure/clock.py +53 -0
- agent_evolve/infrastructure/events/__init__.py +6 -0
- agent_evolve/infrastructure/events/_validation.py +89 -0
- agent_evolve/infrastructure/events/in_memory.py +56 -0
- agent_evolve/infrastructure/events/jsonl.py +193 -0
- agent_evolve/infrastructure/exception_provenance.py +215 -0
- agent_evolve/infrastructure/ids.py +118 -0
- agent_evolve/infrastructure/lineage_codec.py +1836 -0
- agent_evolve/infrastructure/outcome_adaptive_phase_journal.py +170 -0
- agent_evolve/infrastructure/residual_headroom_journal.py +221 -0
- agent_evolve/infrastructure/resource_lease.py +370 -0
- agent_evolve/infrastructure/sanitization/__init__.py +8 -0
- agent_evolve/infrastructure/sanitization/strict_json.py +484 -0
- agent_evolve/infrastructure/sequential_phase_journal.py +170 -0
- agent_evolve/infrastructure/stream_liveness.py +383 -0
- agent_evolve/infrastructure/subprocess_boundary.py +136 -0
- agent_evolve/integrations/__init__.py +1 -0
- agent_evolve/integrations/botorch/__init__.py +28 -0
- agent_evolve/integrations/botorch/finite_qlognehvi.py +190 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_batch.py +155 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_batch_identity.py +20 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_batch_worker.py +55 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_identity.py +22 -0
- agent_evolve/integrations/botorch/finite_qlognehvi_worker.py +55 -0
- agent_evolve/integrations/botorch/subprocess_qlognehvi.py +261 -0
- agent_evolve/integrations/botorch/subprocess_qlognehvi_batch.py +273 -0
- agent_evolve/integrations/completion.py +242 -0
- agent_evolve/integrations/pydantic_ai/__init__.py +441 -0
- agent_evolve/integrations/pydantic_ai/action_forecast.py +1068 -0
- agent_evolve/integrations/pydantic_ai/agentic_generator.py +2308 -0
- agent_evolve/integrations/pydantic_ai/async_generator.py +1604 -0
- agent_evolve/integrations/pydantic_ai/boundary_codec.py +1526 -0
- agent_evolve/integrations/pydantic_ai/calibrated_portfolio_campaign.py +756 -0
- agent_evolve/integrations/pydantic_ai/calibrated_portfolio_selection.py +7537 -0
- agent_evolve/integrations/pydantic_ai/campaign_acquisition.py +609 -0
- agent_evolve/integrations/pydantic_ai/execution_binding.py +138 -0
- agent_evolve/integrations/pydantic_ai/forecast_geometry_action_committee.py +217 -0
- agent_evolve/integrations/pydantic_ai/harness.py +159 -0
- agent_evolve/integrations/pydantic_ai/heterogeneous_model_execution.py +306 -0
- agent_evolve/integrations/pydantic_ai/hierarchical_residual_adaptive_semantic_view.py +179 -0
- agent_evolve/integrations/pydantic_ai/json_schema_dialect.py +108 -0
- agent_evolve/integrations/pydantic_ai/materialized_hierarchical_residual_expert.py +952 -0
- agent_evolve/integrations/pydantic_ai/materialized_portfolio_judge.py +520 -0
- agent_evolve/integrations/pydantic_ai/model_execution_profile.py +659 -0
- agent_evolve/integrations/pydantic_ai/outbound_request_manifest.py +1170 -0
- agent_evolve/integrations/pydantic_ai/portable_residual_consequence_features.py +575 -0
- agent_evolve/integrations/pydantic_ai/portfolio_selection.py +422 -0
- agent_evolve/integrations/pydantic_ai/progress_aware_openrouter.py +416 -0
- agent_evolve/integrations/pydantic_ai/provider_attempt_join.py +1523 -0
- agent_evolve/integrations/pydantic_ai/provider_free_calibrated_runner.py +607 -0
- agent_evolve/integrations/pydantic_ai/queued_runner.py +2634 -0
- agent_evolve/integrations/pydantic_ai/reconciled_residual_reachability.py +1417 -0
- agent_evolve/integrations/pydantic_ai/residual_forecast_geometry.py +445 -0
- agent_evolve/integrations/pydantic_ai/residual_reachability.py +674 -0
- agent_evolve/integrations/pydantic_ai/residual_semantic_cells.py +239 -0
- agent_evolve/integrations/pydantic_ai/sealed_output_replay.py +1068 -0
- agent_evolve/integrations/pydantic_ai/semantic_coverage_residual_portfolio.py +770 -0
- agent_evolve/integrations/pydantic_ai/semantic_decision_replay.py +383 -0
- agent_evolve/integrations/pydantic_ai/support_adaptive_residual_portfolio.py +135 -0
- agent_evolve/integrations/pydantic_ai/trusted_residual_prompt_context.py +143 -0
- agent_evolve/integrations/pydantic_ai/validated_openrouter_model.py +107 -0
- agent_evolve/integrations/pymoo_adapter.py +242 -0
- agent_evolve/policies/__init__.py +17 -0
- agent_evolve/policies/check.py +469 -0
- agent_evolve/policies/emit_scaffold.py +451 -0
- agent_evolve/policies/feedback/__init__.py +37 -0
- agent_evolve/policies/feedback/held_out_asn.py +1325 -0
- agent_evolve/policies/genetic.py +607 -0
- agent_evolve/policies/llm_backoff.py +183 -0
- agent_evolve/policies/llm_chooser.py +226 -0
- agent_evolve/policies/llm_generator.py +1760 -0
- agent_evolve/policies/llm_init.py +267 -0
- agent_evolve/policies/llm_operator.py +109 -0
- agent_evolve/policies/llm_prior.py +194 -0
- agent_evolve/policies/llm_surrogate.py +334 -0
- agent_evolve/policies/measurement_evidence.py +704 -0
- agent_evolve/policies/memory/__init__.py +223 -0
- agent_evolve/policies/memory/balanced_subset_blocks.py +707 -0
- agent_evolve/policies/memory/compatibility_matching.py +593 -0
- agent_evolve/policies/memory/global_falsification.py +1841 -0
- agent_evolve/policies/memory/prompt_shape.py +503 -0
- agent_evolve/policies/memory/randomized_subset.py +714 -0
- agent_evolve/policies/memory/staged_causal.py +1270 -0
- agent_evolve/policies/memory/treatment_compliance.py +759 -0
- agent_evolve/policies/objective_resolution/__init__.py +17 -0
- agent_evolve/policies/objective_resolution/fixed_grid.py +364 -0
- agent_evolve/policies/operator_portfolio.py +407 -0
- agent_evolve/policies/reguidance.py +1133 -0
- agent_evolve/policies/reward/__init__.py +83 -0
- agent_evolve/policies/reward/affine_candidate_consequence.py +156 -0
- agent_evolve/policies/reward/affine_candidate_consequence_3d.py +159 -0
- agent_evolve/policies/reward/affine_hypervolume.py +490 -0
- agent_evolve/policies/reward/affine_hypervolume_3d.py +567 -0
- agent_evolve/policies/reward/contextual_marginal_utility.py +318 -0
- agent_evolve/policies/reward/frozen_archive.py +360 -0
- agent_evolve/policies/reward/frozen_wave_archive.py +368 -0
- agent_evolve/policies/search_state.py +208 -0
- agent_evolve/policies/selection/__init__.py +345 -0
- agent_evolve/policies/selection/acquisition_certified_slate.py +684 -0
- agent_evolve/policies/selection/affine_frontier_context.py +330 -0
- agent_evolve/policies/selection/affine_frontier_target.py +473 -0
- agent_evolve/policies/selection/archive_elite.py +1346 -0
- agent_evolve/policies/selection/calibrated_portfolio_binding.py +640 -0
- agent_evolve/policies/selection/calibrated_slate.py +1394 -0
- agent_evolve/policies/selection/calibrated_slate_codec.py +579 -0
- agent_evolve/policies/selection/common_candidate_pool.py +685 -0
- agent_evolve/policies/selection/diagnostic_sampling.py +319 -0
- agent_evolve/policies/selection/disjoint_pairs.py +479 -0
- agent_evolve/policies/selection/elite_explorer.py +719 -0
- agent_evolve/policies/selection/finite_action.py +187 -0
- agent_evolve/policies/selection/finite_option_prompt_projection.py +377 -0
- agent_evolve/policies/selection/finite_palette_evidence.py +247 -0
- agent_evolve/policies/selection/forecast_calibration.py +922 -0
- agent_evolve/policies/selection/frontier_probe_slate.py +814 -0
- agent_evolve/policies/selection/frozen_archive_pairs.py +762 -0
- agent_evolve/policies/selection/full_support_slate.py +91 -0
- agent_evolve/policies/selection/meaningful_direction.py +240 -0
- agent_evolve/policies/selection/memory_dose_feasibility.py +259 -0
- agent_evolve/policies/selection/model_anchored_slate.py +826 -0
- agent_evolve/policies/selection/phenotype_recourse.py +979 -0
- agent_evolve/policies/selection/proposal_support.py +368 -0
- agent_evolve/policies/selection/random_portfolio.py +254 -0
- agent_evolve/policies/selection/regret_bounded_slate.py +1084 -0
- agent_evolve/policies/selection/residual_frontier.py +463 -0
- agent_evolve/policies/selection/residual_frontier_target.py +605 -0
- agent_evolve/policies/selection/structural_posterior_slate.py +1571 -0
- agent_evolve/policies/selection/target_conditioned_allocator.py +648 -0
- agent_evolve/policies/selection/target_conditioned_features.py +812 -0
- agent_evolve/policies/selection/target_conditioned_prequential.py +1527 -0
- agent_evolve/policies/selection/task_keyed_palette.py +906 -0
- agent_evolve/policies/semantics.py +147 -0
- agent_evolve/policies/structure.py +362 -0
- agent_evolve/policies/structured_output_budget.py +62 -0
- agent_evolve/policies/surrogate.py +696 -0
- agent_evolve/policies/variation/__init__.py +1 -0
- agent_evolve/policies/variation/compositional_finite_catalog.py +426 -0
- agent_evolve/policies/variation/crossover_inheritance.py +575 -0
- agent_evolve/policies/variation/disjoint_recombination.py +611 -0
- agent_evolve/policies/variation/exact_composition_capacity.py +214 -0
- agent_evolve/policies/variation/exact_parent_crossover.py +950 -0
- agent_evolve/policies/variation/multiscale_restart_catalog.py +372 -0
- agent_evolve/policies/variation/source_union_finite_catalog.py +403 -0
- agent_evolve/policies/variation/typed_patch.py +1981 -0
- agent_evolve/policies/weighted_prior.py +394 -0
- agent_evolve/ports/__init__.py +383 -0
- agent_evolve/ports/action_allocation.py +733 -0
- agent_evolve/ports/action_allocation_frame.py +1153 -0
- agent_evolve/ports/action_allocation_frame_commit.py +294 -0
- agent_evolve/ports/action_allocation_frame_commit_v3.py +432 -0
- agent_evolve/ports/action_allocation_frame_v3.py +995 -0
- agent_evolve/ports/action_forecast.py +1568 -0
- agent_evolve/ports/action_metric_projection.py +165 -0
- agent_evolve/ports/agentic_generator.py +1561 -0
- agent_evolve/ports/archive_context.py +136 -0
- agent_evolve/ports/artifact_sanitizer.py +44 -0
- agent_evolve/ports/artifact_store.py +225 -0
- agent_evolve/ports/clock.py +13 -0
- agent_evolve/ports/contextual_search_allocation.py +827 -0
- agent_evolve/ports/decision_metric_projection.py +258 -0
- agent_evolve/ports/event_store.py +55 -0
- agent_evolve/ports/executable_hypothesis.py +557 -0
- agent_evolve/ports/finite_acquisition.py +377 -0
- agent_evolve/ports/finite_acquisition_batch.py +296 -0
- agent_evolve/ports/finite_acquisition_batch_json.py +164 -0
- agent_evolve/ports/finite_acquisition_json.py +247 -0
- agent_evolve/ports/finite_acquisition_space.py +168 -0
- agent_evolve/ports/finite_action_selection.py +348 -0
- agent_evolve/ports/finite_action_set.py +256 -0
- agent_evolve/ports/frontier_target.py +396 -0
- agent_evolve/ports/generation_failure.py +43 -0
- agent_evolve/ports/hard_feasibility.py +233 -0
- agent_evolve/ports/id_factory.py +34 -0
- agent_evolve/ports/llm_task_queue.py +93 -0
- agent_evolve/ports/objective_resolution.py +419 -0
- agent_evolve/ports/paired_allocation_comparison.py +401 -0
- agent_evolve/ports/paired_block_schedule.py +475 -0
- agent_evolve/ports/parent_measurement.py +336 -0
- agent_evolve/ports/portfolio_memory_dose.py +643 -0
- agent_evolve/ports/portfolio_selection.py +3169 -0
- agent_evolve/ports/postcommit_rank_authority.py +467 -0
- agent_evolve/ports/presented_action_evidence.py +794 -0
- agent_evolve/ports/resource_lease.py +162 -0
- agent_evolve/ports/structured_generator.py +734 -0
- agent_evolve/ports/structured_output_budget.py +120 -0
- agent_evolve/ports/subprocess_boundary.py +138 -0
- agent_evolve/ports/treatment_assignment.py +466 -0
- agent_evolve/ports/variation_catalog.py +76 -0
- agent_evolve/ports/variation_source.py +226 -0
- agent_evolve/proposal_mode.py +157 -0
- agent_evolve/proposers/__init__.py +10 -0
- agent_evolve/proposers/random_proposer.py +188 -0
- agent_evolve/provider_accounting.py +163 -0
- agent_evolve/py.typed +0 -0
- agent_evolve/reference_method.py +1570 -0
- agent_evolve/session/__init__.py +11 -0
- agent_evolve/session/authorship.py +864 -0
- agent_evolve/session/evaluate.py +236 -0
- agent_evolve/session/fidelity.py +237 -0
- agent_evolve/session/genetic_loop.py +742 -0
- agent_evolve/session/loop.py +803 -0
- agent_evolve/session/screening.py +671 -0
- agent_evolve/settings.py +376 -0
- agent_evolve/workload_kit.py +368 -0
- agent_evolve/workload_prompt.py +398 -0
- agentevolve_optimizer-0.5.0.dist-info/METADATA +599 -0
- agentevolve_optimizer-0.5.0.dist-info/RECORD +414 -0
- agentevolve_optimizer-0.5.0.dist-info/WHEEL +5 -0
- agentevolve_optimizer-0.5.0.dist-info/entry_points.txt +2 -0
- agentevolve_optimizer-0.5.0.dist-info/licenses/LICENSE +21 -0
- agentevolve_optimizer-0.5.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,2634 @@
|
|
|
1
|
+
"""Queue-owned retry composition for one-attempt structured generation.
|
|
2
|
+
|
|
3
|
+
The Pydantic-AI adapter remains a one-attempt executor. This module gives the
|
|
4
|
+
application queue sole ownership of admission bounds, timeouts, retries,
|
|
5
|
+
backoff, and sleeps, then exposes the callable expected by the high-level
|
|
6
|
+
agentic adapter.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import asyncio
|
|
12
|
+
import hashlib
|
|
13
|
+
import json
|
|
14
|
+
import math
|
|
15
|
+
import re
|
|
16
|
+
from collections.abc import Callable, Mapping
|
|
17
|
+
from dataclasses import dataclass, replace
|
|
18
|
+
from decimal import Decimal, ROUND_CEILING
|
|
19
|
+
from enum import Enum
|
|
20
|
+
from functools import partial
|
|
21
|
+
from random import SystemRandom
|
|
22
|
+
from typing import Any, Generic, Protocol, cast, runtime_checkable
|
|
23
|
+
|
|
24
|
+
from pydantic import BaseModel
|
|
25
|
+
|
|
26
|
+
from agent_evolve.application.llm_task_queue import (
|
|
27
|
+
AsyncLLMTaskQueue,
|
|
28
|
+
LLMTaskQueueClosedError,
|
|
29
|
+
)
|
|
30
|
+
from agent_evolve.domain.ids import LLMCallId, ProviderAttemptId
|
|
31
|
+
from agent_evolve.domain.llm_task_queue import (
|
|
32
|
+
MAX_ATTEMPTS,
|
|
33
|
+
NANOSECONDS_PER_SECOND,
|
|
34
|
+
AttemptRequestEvidence,
|
|
35
|
+
AttemptRequestVariant,
|
|
36
|
+
LLMAttemptContext,
|
|
37
|
+
LLMTask,
|
|
38
|
+
LLMTaskOutcome,
|
|
39
|
+
PartitionedRetryBudget,
|
|
40
|
+
QueueSnapshot,
|
|
41
|
+
RetryAfter,
|
|
42
|
+
RetryAfterSource,
|
|
43
|
+
RetryClassification,
|
|
44
|
+
RetryDisposition,
|
|
45
|
+
RetryReason,
|
|
46
|
+
SanitizedAttemptFailure,
|
|
47
|
+
StructuredOutputFailureMode,
|
|
48
|
+
TaskOutcomeStatus,
|
|
49
|
+
TaskTelemetry,
|
|
50
|
+
ValidationIssueCategory,
|
|
51
|
+
ValidationIssueReasonCode,
|
|
52
|
+
)
|
|
53
|
+
from agent_evolve.infrastructure.asyncio_runtime import (
|
|
54
|
+
AsyncioRuntime,
|
|
55
|
+
TransportAbortedTimeoutError,
|
|
56
|
+
)
|
|
57
|
+
from agent_evolve.infrastructure.clock import SystemClock
|
|
58
|
+
from agent_evolve.integrations.pydantic_ai.agentic_generator import (
|
|
59
|
+
AttemptedStructuredGenerationResponse,
|
|
60
|
+
)
|
|
61
|
+
from agent_evolve.integrations.pydantic_ai.async_generator import (
|
|
62
|
+
PydanticAIStructuredGenerator,
|
|
63
|
+
)
|
|
64
|
+
from agent_evolve.policies.llm_backoff import (
|
|
65
|
+
ExponentialBackoff,
|
|
66
|
+
FullJitter,
|
|
67
|
+
JitterPolicy,
|
|
68
|
+
RandomRange,
|
|
69
|
+
)
|
|
70
|
+
from agent_evolve.ports.llm_task_queue import (
|
|
71
|
+
AsyncRuntime,
|
|
72
|
+
BackoffPolicy,
|
|
73
|
+
PreparedLLMAttempt,
|
|
74
|
+
RetryClassifier,
|
|
75
|
+
)
|
|
76
|
+
from agent_evolve.ports.generation_failure import GenerationFailureDisposition
|
|
77
|
+
from agent_evolve.ports.structured_generator import (
|
|
78
|
+
GenerationFailureKind,
|
|
79
|
+
IDENTITY_PROMPT_RENDERER_DEFINITION_SHA256,
|
|
80
|
+
IDENTITY_PROMPT_RENDERER_ID,
|
|
81
|
+
IDENTITY_PROMPT_RENDERER_REVISION,
|
|
82
|
+
MAX_OUTPUT_TOKENS,
|
|
83
|
+
MAX_PROMPT_UTF8_BYTES,
|
|
84
|
+
OutputT,
|
|
85
|
+
StructuredGenerationError,
|
|
86
|
+
StructuredGenerationRequest,
|
|
87
|
+
StructuredGenerationResponse,
|
|
88
|
+
StructuredGenerator,
|
|
89
|
+
StructuredPromptLineage,
|
|
90
|
+
StructuredStreamCleanupTimeoutError,
|
|
91
|
+
StructuredStreamTimeoutError,
|
|
92
|
+
StructuredStreamTimeoutPhase,
|
|
93
|
+
identity_prompt_lineage,
|
|
94
|
+
)
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
DEFAULT_MAX_IN_FLIGHT = 8
|
|
98
|
+
DEFAULT_MAX_PENDING = 64
|
|
99
|
+
DEFAULT_MAX_ATTEMPTS = 3
|
|
100
|
+
DEFAULT_ATTEMPT_TIMEOUT_NS = 90 * NANOSECONDS_PER_SECOND
|
|
101
|
+
DEFAULT_BASE_BACKOFF_NS = NANOSECONDS_PER_SECOND // 2
|
|
102
|
+
DEFAULT_MAX_BACKOFF_NS = 30 * NANOSECONDS_PER_SECOND
|
|
103
|
+
_MAX_RETRY_AFTER_NS = 2**63 - 1
|
|
104
|
+
_PROVIDER_ATTEMPT_ID_DOMAIN = b"agent-evolve:provider-attempt-id:v1\x00"
|
|
105
|
+
_STRUCTURED_REQUEST_EVIDENCE_DOMAIN = b"agent-evolve:structured-request-evidence:v2\x00"
|
|
106
|
+
_STRUCTURED_OUTPUT_EVIDENCE_DOMAIN = b"agent-evolve:structured-output-evidence:v1\x00"
|
|
107
|
+
_POLICY_ID = re.compile(r"^[a-z][a-z0-9_]{0,63}$")
|
|
108
|
+
_LOWER_SHA256 = re.compile(r"^[0-9a-f]{64}$")
|
|
109
|
+
_EVIDENCE_OPERATION = re.compile(r"^[a-z][a-z0-9_.-]{0,95}$")
|
|
110
|
+
_EVIDENCE_TOOL = re.compile(r"^[A-Za-z][A-Za-z0-9_-]{0,63}$")
|
|
111
|
+
MAX_STRUCTURED_OUTPUT_SCHEMA_UTF8_BYTES = 1_048_576
|
|
112
|
+
MAX_STRUCTURED_OUTPUT_EVIDENCE_UTF8_BYTES = 4_194_304
|
|
113
|
+
STRUCTURED_REQUEST_EVIDENCE_SCHEMA_VERSION = 2
|
|
114
|
+
STRUCTURED_OUTPUT_EVIDENCE_SCHEMA_VERSION = 1
|
|
115
|
+
STRUCTURED_GENERATION_OUTCOME_SCHEMA_VERSION = 8
|
|
116
|
+
SUPPORTED_STRUCTURED_GENERATION_OUTCOME_SCHEMA_VERSIONS = frozenset({5, 6, 7, 8})
|
|
117
|
+
_STRUCTURED_REQUEST_EVIDENCE_FIELDS = frozenset(
|
|
118
|
+
{
|
|
119
|
+
"schema_version",
|
|
120
|
+
"call_id",
|
|
121
|
+
"operation",
|
|
122
|
+
"prompt_sha256",
|
|
123
|
+
"wire_prompt_sha256",
|
|
124
|
+
"prompt_utf8_bytes",
|
|
125
|
+
"semantic_prompt_sha256",
|
|
126
|
+
"prompt_renderer_id",
|
|
127
|
+
"prompt_renderer_revision",
|
|
128
|
+
"prompt_renderer_definition_sha256",
|
|
129
|
+
"output_tool_name",
|
|
130
|
+
"output_type",
|
|
131
|
+
"output_schema",
|
|
132
|
+
"output_schema_sha256",
|
|
133
|
+
"output_schema_utf8_bytes",
|
|
134
|
+
"max_output_tokens",
|
|
135
|
+
"temperature_hex",
|
|
136
|
+
"request_evidence_sha256",
|
|
137
|
+
}
|
|
138
|
+
)
|
|
139
|
+
_STRUCTURED_OUTPUT_EVIDENCE_FIELDS = frozenset(
|
|
140
|
+
{
|
|
141
|
+
"schema_version",
|
|
142
|
+
"call_id",
|
|
143
|
+
"operation",
|
|
144
|
+
"provider_response_id",
|
|
145
|
+
"request_evidence_sha256",
|
|
146
|
+
"output_tool_name",
|
|
147
|
+
"output_schema_sha256",
|
|
148
|
+
"typed_output",
|
|
149
|
+
"typed_output_sha256",
|
|
150
|
+
"typed_output_utf8_bytes",
|
|
151
|
+
"output_evidence_sha256",
|
|
152
|
+
}
|
|
153
|
+
)
|
|
154
|
+
SCHEMA_REPAIR_POLICY_ID = "structured_output_schema_repair"
|
|
155
|
+
SCHEMA_REPAIR_POLICY_VERSION = 4
|
|
156
|
+
SCHEMA_REPAIR_PROMPT_RENDERER_ID = "agent_evolve.schema_repair_prompt"
|
|
157
|
+
SCHEMA_REPAIR_PROMPT_RENDERER_REVISION = "schema_repair_v4"
|
|
158
|
+
MAX_SCHEMA_REPAIR_SUFFIX_UTF8_BYTES = 24_576
|
|
159
|
+
MAX_SCHEMA_REPAIR_SCHEMA_NODES = 4_096
|
|
160
|
+
MAX_SCHEMA_REPAIR_REQUIRED_PATHS = 256
|
|
161
|
+
_SCHEMA_REPAIR_TEMPLATE = (
|
|
162
|
+
"\n\nSTRUCTURED_OUTPUT_SCHEMA_REPAIR_V{policy_version}\n"
|
|
163
|
+
"The previous provider response did not satisfy the typed output contract. "
|
|
164
|
+
"Failure mode: {failure_mode}. Repair pass: {repair_pass}.\n"
|
|
165
|
+
"Schema-required field paths from the trusted local output contract "
|
|
166
|
+
"(JSON Pointer; '*' marks each emitted collection item): "
|
|
167
|
+
"{required_paths_json}\n"
|
|
168
|
+
"Include every applicable path above. "
|
|
169
|
+
"{issue_block}"
|
|
170
|
+
"{literal_constraint_block}"
|
|
171
|
+
"Call the {output_tool_name} output tool exactly once. Emit only fields "
|
|
172
|
+
"declared by its schema, include every required field, and use exact schema "
|
|
173
|
+
"literals, enums, and types. Do not emit commentary outside the tool call."
|
|
174
|
+
"{completion_guidance}{escalation_guidance}"
|
|
175
|
+
)
|
|
176
|
+
_SEMANTIC_REPAIR_GUIDANCE: dict[ValidationIssueReasonCode, str] = {
|
|
177
|
+
ValidationIssueReasonCode.DUPLICATE_FINITE_OPTIONS: (
|
|
178
|
+
"Correction: every proposed finite option ID must be distinct."
|
|
179
|
+
),
|
|
180
|
+
ValidationIssueReasonCode.FINITE_OPTION_OUT_OF_CONTRACT: (
|
|
181
|
+
"Correction: every proposed option_id must exactly match one option_id "
|
|
182
|
+
"from the request's sealed ordered_options list."
|
|
183
|
+
),
|
|
184
|
+
ValidationIssueReasonCode.ASSIGNED_MEMORY_CARD_OMITTED: (
|
|
185
|
+
"Correction: across the complete proposal, include every prospectively "
|
|
186
|
+
"assigned memory-card key in at least one member's "
|
|
187
|
+
"supporting_card_keys; also obey the supplied compatibility and dose "
|
|
188
|
+
"bounds exactly."
|
|
189
|
+
),
|
|
190
|
+
ValidationIssueReasonCode.PROPOSAL_SUPPORT_OPTION_OMITTED: (
|
|
191
|
+
"Correction: include every engine-reserved proposal-support option in "
|
|
192
|
+
"the complete proposal; copy each reserved option ID exactly from the "
|
|
193
|
+
"trusted request and keep all proposal members distinct."
|
|
194
|
+
),
|
|
195
|
+
ValidationIssueReasonCode.NO_FEASIBLE_DISJOINT_PORTFOLIO: (
|
|
196
|
+
"Correction: the complete proposal must contain a subset satisfying "
|
|
197
|
+
"the supplied evaluation-size, pairwise changed-path, and distinct-family "
|
|
198
|
+
"constraints."
|
|
199
|
+
),
|
|
200
|
+
ValidationIssueReasonCode.PORTFOLIO_MEMORY_DOSE_VIOLATION: (
|
|
201
|
+
"Correction: obey every supplied memory-dose bound, cite only "
|
|
202
|
+
"card-compatible options, include every assigned card, and preserve the "
|
|
203
|
+
"required unattributed-member count."
|
|
204
|
+
),
|
|
205
|
+
ValidationIssueReasonCode.REFLECTION_METRIC_CONTRACT_VIOLATION: (
|
|
206
|
+
"Correction: in every insight, emit each required metric exactly once "
|
|
207
|
+
"and emit no other metric."
|
|
208
|
+
),
|
|
209
|
+
ValidationIssueReasonCode.REFLECTION_ACTION_CONTRACT_VIOLATION: (
|
|
210
|
+
"Correction: use at least one allowed recommended option ID and family; "
|
|
211
|
+
"use only request-listed values, remove duplicates, and keep every "
|
|
212
|
+
"recommended ID, family, affected path, and capability mutually "
|
|
213
|
+
"consistent with the cited observed action."
|
|
214
|
+
),
|
|
215
|
+
ValidationIssueReasonCode.REFLECTION_SEMANTIC_CONTRACT_VIOLATION: (
|
|
216
|
+
"Correction: use only the request-listed insight kind, consumer scope, "
|
|
217
|
+
"affected path, and factor capability values; required set-like arrays "
|
|
218
|
+
"must be nonempty where specified and contain no duplicates."
|
|
219
|
+
),
|
|
220
|
+
ValidationIssueReasonCode.REFLECTION_DIRECTION_OR_ANCHOR_VIOLATION: (
|
|
221
|
+
"Correction: every metric prediction must use an adjudicable non-unknown "
|
|
222
|
+
"direction and an explicitly allowed comparison anchor; supply a source "
|
|
223
|
+
"role only when that anchor kind requires one."
|
|
224
|
+
),
|
|
225
|
+
ValidationIssueReasonCode.RESIDUAL_RADIUS_CONTRACT_VIOLATION: (
|
|
226
|
+
"Correction: every residual member must contain exactly an allowed "
|
|
227
|
+
"number of component_option_ids."
|
|
228
|
+
),
|
|
229
|
+
ValidationIssueReasonCode.RESIDUAL_OPTION_CONTRACT_VIOLATION: (
|
|
230
|
+
"Correction: within each residual member, use distinct option IDs that "
|
|
231
|
+
"are all available for its selected parent; a two-option member must "
|
|
232
|
+
"copy one of that parent's declared safe disjoint pairs."
|
|
233
|
+
),
|
|
234
|
+
ValidationIssueReasonCode.RESIDUAL_METRIC_CONTRACT_VIOLATION: (
|
|
235
|
+
"Correction: within every residual member, emit each required metric "
|
|
236
|
+
"exactly once and emit no other metric."
|
|
237
|
+
),
|
|
238
|
+
ValidationIssueReasonCode.RESIDUAL_QUANTILE_ORDER_VIOLATION: (
|
|
239
|
+
"Correction: every metric forecast must contain finite raw deltas in "
|
|
240
|
+
"nondecreasing order: p10_delta <= p50_delta <= p90_delta."
|
|
241
|
+
),
|
|
242
|
+
ValidationIssueReasonCode.RESIDUAL_PLAN_DIVERSITY_VIOLATION: (
|
|
243
|
+
"Correction: residual members must be distinct parent-relative plans "
|
|
244
|
+
"and collectively cover at least the requested number of distinct "
|
|
245
|
+
"parents."
|
|
246
|
+
),
|
|
247
|
+
}
|
|
248
|
+
_DEFAULT_SEMANTIC_REPAIR_GUIDANCE = (
|
|
249
|
+
"Correction: rebuild the complete typed output and satisfy the named "
|
|
250
|
+
"trusted semantic constraint."
|
|
251
|
+
)
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
def _semantic_repair_guidance_record() -> dict[str, object]:
|
|
255
|
+
"""Return the complete deterministic guidance contract.
|
|
256
|
+
|
|
257
|
+
Validator reason codes and prompt guidance evolve in separate modules. A
|
|
258
|
+
total, content-addressed record prevents either enum drift or wording drift
|
|
259
|
+
from changing retry behavior under an unchanged experiment identity.
|
|
260
|
+
"""
|
|
261
|
+
|
|
262
|
+
missing = set(ValidationIssueReasonCode).difference(_SEMANTIC_REPAIR_GUIDANCE)
|
|
263
|
+
extra = set(_SEMANTIC_REPAIR_GUIDANCE).difference(ValidationIssueReasonCode)
|
|
264
|
+
if missing or extra:
|
|
265
|
+
raise ValueError(
|
|
266
|
+
"semantic repair guidance must cover every validation reason code "
|
|
267
|
+
"exactly"
|
|
268
|
+
)
|
|
269
|
+
return {
|
|
270
|
+
"default_guidance": _DEFAULT_SEMANTIC_REPAIR_GUIDANCE,
|
|
271
|
+
"reason_guidance": {
|
|
272
|
+
reason.value: _SEMANTIC_REPAIR_GUIDANCE[reason]
|
|
273
|
+
for reason in sorted(
|
|
274
|
+
ValidationIssueReasonCode,
|
|
275
|
+
key=lambda item: item.value,
|
|
276
|
+
)
|
|
277
|
+
},
|
|
278
|
+
}
|
|
279
|
+
|
|
280
|
+
|
|
281
|
+
@dataclass(frozen=True, slots=True)
|
|
282
|
+
class SchemaRepairPolicyManifest:
|
|
283
|
+
"""Immutable, self-authenticating schema-repair experiment contract."""
|
|
284
|
+
|
|
285
|
+
policy_id: str
|
|
286
|
+
policy_version: int
|
|
287
|
+
max_suffix_utf8_bytes: int
|
|
288
|
+
max_schema_nodes: int
|
|
289
|
+
max_required_paths: int
|
|
290
|
+
template_sha256: str
|
|
291
|
+
semantic_guidance_sha256: str
|
|
292
|
+
policy_sha256: str
|
|
293
|
+
|
|
294
|
+
def __post_init__(self) -> None:
|
|
295
|
+
if (
|
|
296
|
+
type(self.policy_id) is not str
|
|
297
|
+
or _POLICY_ID.fullmatch(self.policy_id) is None
|
|
298
|
+
):
|
|
299
|
+
raise ValueError("policy_id must use the closed lowercase token grammar")
|
|
300
|
+
if type(self.policy_version) is not int or self.policy_version < 1:
|
|
301
|
+
raise ValueError("policy_version must be a positive integer")
|
|
302
|
+
if (
|
|
303
|
+
type(self.max_suffix_utf8_bytes) is not int
|
|
304
|
+
or not 1 <= self.max_suffix_utf8_bytes <= MAX_PROMPT_UTF8_BYTES
|
|
305
|
+
):
|
|
306
|
+
raise ValueError("max_suffix_utf8_bytes is outside the prompt boundary")
|
|
307
|
+
if type(self.max_schema_nodes) is not int or self.max_schema_nodes < 1:
|
|
308
|
+
raise ValueError("max_schema_nodes must be a positive integer")
|
|
309
|
+
if type(self.max_required_paths) is not int or self.max_required_paths < 1:
|
|
310
|
+
raise ValueError("max_required_paths must be a positive integer")
|
|
311
|
+
for name, value in (
|
|
312
|
+
("template_sha256", self.template_sha256),
|
|
313
|
+
("semantic_guidance_sha256", self.semantic_guidance_sha256),
|
|
314
|
+
("policy_sha256", self.policy_sha256),
|
|
315
|
+
):
|
|
316
|
+
if type(value) is not str or _LOWER_SHA256.fullmatch(value) is None:
|
|
317
|
+
raise ValueError(f"{name} must be a lowercase SHA-256 digest")
|
|
318
|
+
expected = hashlib.sha256(
|
|
319
|
+
json.dumps(
|
|
320
|
+
self._policy_record(),
|
|
321
|
+
allow_nan=False,
|
|
322
|
+
ensure_ascii=True,
|
|
323
|
+
separators=(",", ":"),
|
|
324
|
+
sort_keys=True,
|
|
325
|
+
).encode("ascii")
|
|
326
|
+
).hexdigest()
|
|
327
|
+
if self.policy_sha256 != expected:
|
|
328
|
+
raise ValueError("policy_sha256 does not authenticate the policy fields")
|
|
329
|
+
|
|
330
|
+
def _policy_record(self) -> dict[str, object]:
|
|
331
|
+
return {
|
|
332
|
+
"max_required_paths": self.max_required_paths,
|
|
333
|
+
"max_schema_nodes": self.max_schema_nodes,
|
|
334
|
+
"max_suffix_utf8_bytes": self.max_suffix_utf8_bytes,
|
|
335
|
+
"policy_id": self.policy_id,
|
|
336
|
+
"policy_version": self.policy_version,
|
|
337
|
+
"semantic_guidance_sha256": self.semantic_guidance_sha256,
|
|
338
|
+
"template_sha256": self.template_sha256,
|
|
339
|
+
}
|
|
340
|
+
|
|
341
|
+
def to_trace_record(self) -> dict[str, object]:
|
|
342
|
+
"""Return the complete JSON-safe contract for a launch manifest."""
|
|
343
|
+
|
|
344
|
+
return {
|
|
345
|
+
**self._policy_record(),
|
|
346
|
+
"policy_sha256": self.policy_sha256,
|
|
347
|
+
}
|
|
348
|
+
|
|
349
|
+
|
|
350
|
+
def _schema_repair_policy_manifest() -> SchemaRepairPolicyManifest:
|
|
351
|
+
template_sha256 = hashlib.sha256(
|
|
352
|
+
_SCHEMA_REPAIR_TEMPLATE.encode("utf-8")
|
|
353
|
+
).hexdigest()
|
|
354
|
+
semantic_guidance_sha256 = hashlib.sha256(
|
|
355
|
+
json.dumps(
|
|
356
|
+
_semantic_repair_guidance_record(),
|
|
357
|
+
allow_nan=False,
|
|
358
|
+
ensure_ascii=True,
|
|
359
|
+
separators=(",", ":"),
|
|
360
|
+
sort_keys=True,
|
|
361
|
+
).encode("ascii")
|
|
362
|
+
).hexdigest()
|
|
363
|
+
policy_record = {
|
|
364
|
+
"max_required_paths": MAX_SCHEMA_REPAIR_REQUIRED_PATHS,
|
|
365
|
+
"max_schema_nodes": MAX_SCHEMA_REPAIR_SCHEMA_NODES,
|
|
366
|
+
"max_suffix_utf8_bytes": MAX_SCHEMA_REPAIR_SUFFIX_UTF8_BYTES,
|
|
367
|
+
"policy_id": SCHEMA_REPAIR_POLICY_ID,
|
|
368
|
+
"policy_version": SCHEMA_REPAIR_POLICY_VERSION,
|
|
369
|
+
"semantic_guidance_sha256": semantic_guidance_sha256,
|
|
370
|
+
"template_sha256": template_sha256,
|
|
371
|
+
}
|
|
372
|
+
policy_sha256 = hashlib.sha256(
|
|
373
|
+
json.dumps(
|
|
374
|
+
policy_record,
|
|
375
|
+
allow_nan=False,
|
|
376
|
+
ensure_ascii=True,
|
|
377
|
+
separators=(",", ":"),
|
|
378
|
+
sort_keys=True,
|
|
379
|
+
).encode("ascii")
|
|
380
|
+
).hexdigest()
|
|
381
|
+
return SchemaRepairPolicyManifest(
|
|
382
|
+
**policy_record,
|
|
383
|
+
policy_sha256=policy_sha256,
|
|
384
|
+
)
|
|
385
|
+
|
|
386
|
+
|
|
387
|
+
SCHEMA_REPAIR_POLICY_MANIFEST = _schema_repair_policy_manifest()
|
|
388
|
+
OutcomeSink = Callable[
|
|
389
|
+
[LLMTaskOutcome[StructuredGenerationResponse[Any]]],
|
|
390
|
+
None,
|
|
391
|
+
]
|
|
392
|
+
StructuredRequestEvidenceSink = Callable[[dict[str, object]], None]
|
|
393
|
+
StructuredOutputEvidenceSink = Callable[[dict[str, object]], None]
|
|
394
|
+
|
|
395
|
+
|
|
396
|
+
class OutcomePublicationPolicy(str, Enum):
|
|
397
|
+
"""Control whether terminal-outcome publication is advisory or required.
|
|
398
|
+
|
|
399
|
+
``REQUIRED`` makes publication a synchronous fail-closed boundary: no
|
|
400
|
+
successful response is returned to downstream validation or experiment
|
|
401
|
+
policy unless the sink returns normally. Actual durability remains the
|
|
402
|
+
sink's responsibility (for example, flush and fsync before returning).
|
|
403
|
+
Publication failure never causes a provider retry because the queue has
|
|
404
|
+
already reached one terminal logical outcome.
|
|
405
|
+
"""
|
|
406
|
+
|
|
407
|
+
BEST_EFFORT = "best_effort"
|
|
408
|
+
REQUIRED = "required"
|
|
409
|
+
|
|
410
|
+
|
|
411
|
+
class StructuredEvidencePublicationPolicy(str, Enum):
|
|
412
|
+
"""Control publication of opt-in request/output content evidence.
|
|
413
|
+
|
|
414
|
+
These records intentionally cross the privacy boundary that the sanitized
|
|
415
|
+
terminal-outcome projection does not: request evidence authenticates the
|
|
416
|
+
exact *wire* prompt and output contract, while successful-output evidence
|
|
417
|
+
contains the canonical typed output itself. Callers must opt in by supplying
|
|
418
|
+
both sinks. ``REQUIRED`` makes both synchronous durability barriers.
|
|
419
|
+
"""
|
|
420
|
+
|
|
421
|
+
BEST_EFFORT = "best_effort"
|
|
422
|
+
REQUIRED = "required"
|
|
423
|
+
|
|
424
|
+
|
|
425
|
+
class StructuredEvidencePublicationStage(str, Enum):
|
|
426
|
+
REQUEST = "request"
|
|
427
|
+
OUTPUT = "output"
|
|
428
|
+
|
|
429
|
+
|
|
430
|
+
def _canonical_evidence_bytes(value: object) -> bytes:
|
|
431
|
+
return json.dumps(
|
|
432
|
+
value,
|
|
433
|
+
ensure_ascii=True,
|
|
434
|
+
allow_nan=False,
|
|
435
|
+
separators=(",", ":"),
|
|
436
|
+
sort_keys=True,
|
|
437
|
+
).encode("ascii")
|
|
438
|
+
|
|
439
|
+
|
|
440
|
+
def _output_schema_record(
|
|
441
|
+
output_type: type[Any],
|
|
442
|
+
) -> tuple[dict[str, object], bytes, str]:
|
|
443
|
+
try:
|
|
444
|
+
schema = output_type.model_json_schema(mode="validation")
|
|
445
|
+
except Exception as exc:
|
|
446
|
+
raise TypeError("structured output type cannot render a JSON schema") from exc
|
|
447
|
+
if type(schema) is not dict:
|
|
448
|
+
raise TypeError("structured output schema must be an exact object")
|
|
449
|
+
schema_bytes = _canonical_evidence_bytes(schema)
|
|
450
|
+
if len(schema_bytes) > MAX_STRUCTURED_OUTPUT_SCHEMA_UTF8_BYTES:
|
|
451
|
+
raise ValueError("structured output schema exceeds the evidence bound")
|
|
452
|
+
return schema, schema_bytes, hashlib.sha256(schema_bytes).hexdigest()
|
|
453
|
+
|
|
454
|
+
|
|
455
|
+
def structured_generation_request_evidence_record(
|
|
456
|
+
request: StructuredGenerationRequest[Any],
|
|
457
|
+
) -> dict[str, object]:
|
|
458
|
+
"""Authenticate one exact prequeue wire request without retaining its prompt.
|
|
459
|
+
|
|
460
|
+
The high-level agentic adapter may render a semantic prompt into a different
|
|
461
|
+
provider-facing prompt (for example by appending a reflection wire-contract
|
|
462
|
+
note). This boundary is downstream of that rendering, so ``prompt_sha256``
|
|
463
|
+
names the bytes actually submitted to the queue rather than an upstream
|
|
464
|
+
semantic-prompt commitment.
|
|
465
|
+
"""
|
|
466
|
+
|
|
467
|
+
if type(request) is not StructuredGenerationRequest:
|
|
468
|
+
raise TypeError("request must be an exact StructuredGenerationRequest")
|
|
469
|
+
StructuredGenerationRequest.__post_init__(request)
|
|
470
|
+
schema, schema_bytes, schema_sha256 = _output_schema_record(request.output_type)
|
|
471
|
+
output_type = request.output_type
|
|
472
|
+
prompt_bytes = request.prompt.encode("utf-8", errors="strict")
|
|
473
|
+
wire_prompt_sha256 = hashlib.sha256(prompt_bytes).hexdigest()
|
|
474
|
+
prompt_lineage = request.prompt_lineage
|
|
475
|
+
record: dict[str, object] = {
|
|
476
|
+
"schema_version": STRUCTURED_REQUEST_EVIDENCE_SCHEMA_VERSION,
|
|
477
|
+
"call_id": request.call_id.value,
|
|
478
|
+
"operation": request.operation,
|
|
479
|
+
# ``prompt_sha256`` is retained as an unambiguous compatibility alias
|
|
480
|
+
# for existing journal consumers; new consumers should use the
|
|
481
|
+
# explicitly named wire-prompt field.
|
|
482
|
+
"prompt_sha256": wire_prompt_sha256,
|
|
483
|
+
"wire_prompt_sha256": wire_prompt_sha256,
|
|
484
|
+
"prompt_utf8_bytes": len(prompt_bytes),
|
|
485
|
+
"semantic_prompt_sha256": (
|
|
486
|
+
None if prompt_lineage is None else prompt_lineage.semantic_prompt_sha256
|
|
487
|
+
),
|
|
488
|
+
"prompt_renderer_id": (
|
|
489
|
+
None if prompt_lineage is None else prompt_lineage.renderer_id
|
|
490
|
+
),
|
|
491
|
+
"prompt_renderer_revision": (
|
|
492
|
+
None if prompt_lineage is None else prompt_lineage.renderer_revision
|
|
493
|
+
),
|
|
494
|
+
"prompt_renderer_definition_sha256": (
|
|
495
|
+
None
|
|
496
|
+
if prompt_lineage is None
|
|
497
|
+
else prompt_lineage.renderer_definition_sha256
|
|
498
|
+
),
|
|
499
|
+
"output_tool_name": request.output_tool_name,
|
|
500
|
+
"output_type": {
|
|
501
|
+
"module": output_type.__module__,
|
|
502
|
+
"qualname": output_type.__qualname__,
|
|
503
|
+
},
|
|
504
|
+
"output_schema": schema,
|
|
505
|
+
"output_schema_sha256": schema_sha256,
|
|
506
|
+
"output_schema_utf8_bytes": len(schema_bytes),
|
|
507
|
+
"max_output_tokens": request.max_output_tokens,
|
|
508
|
+
"temperature_hex": (
|
|
509
|
+
None if request.temperature is None else float(request.temperature).hex()
|
|
510
|
+
),
|
|
511
|
+
}
|
|
512
|
+
record["request_evidence_sha256"] = hashlib.sha256(
|
|
513
|
+
_STRUCTURED_REQUEST_EVIDENCE_DOMAIN + _canonical_evidence_bytes(record)
|
|
514
|
+
).hexdigest()
|
|
515
|
+
return record
|
|
516
|
+
|
|
517
|
+
|
|
518
|
+
def structured_generation_output_evidence_record(
|
|
519
|
+
request: StructuredGenerationRequest[Any],
|
|
520
|
+
outcome: LLMTaskOutcome[StructuredGenerationResponse[Any]],
|
|
521
|
+
*,
|
|
522
|
+
request_evidence: dict[str, object] | None = None,
|
|
523
|
+
) -> dict[str, object]:
|
|
524
|
+
"""Retain one bounded canonical typed output before downstream validation."""
|
|
525
|
+
|
|
526
|
+
if type(request) is not StructuredGenerationRequest:
|
|
527
|
+
raise TypeError("request must be an exact StructuredGenerationRequest")
|
|
528
|
+
StructuredGenerationRequest.__post_init__(request)
|
|
529
|
+
if type(outcome) is not LLMTaskOutcome:
|
|
530
|
+
raise TypeError("outcome must be an exact LLMTaskOutcome")
|
|
531
|
+
LLMTaskOutcome.__post_init__(outcome)
|
|
532
|
+
if outcome.status is not TaskOutcomeStatus.SUCCEEDED:
|
|
533
|
+
raise ValueError("typed output evidence requires a successful outcome")
|
|
534
|
+
if outcome.telemetry.task_id != request.call_id.value:
|
|
535
|
+
raise ValueError("request and successful outcome call identities differ")
|
|
536
|
+
response = outcome.response
|
|
537
|
+
if type(response) is not StructuredGenerationResponse:
|
|
538
|
+
raise TypeError("successful outcome has no structured response")
|
|
539
|
+
StructuredGenerationResponse.__post_init__(response)
|
|
540
|
+
if type(response.value) is not request.output_type or not isinstance(
|
|
541
|
+
response.value, BaseModel
|
|
542
|
+
):
|
|
543
|
+
raise TypeError("successful typed output differs from its output contract")
|
|
544
|
+
expected_request = structured_generation_request_evidence_record(request)
|
|
545
|
+
if request_evidence is None:
|
|
546
|
+
request_record = expected_request
|
|
547
|
+
else:
|
|
548
|
+
if type(request_evidence) is not dict or request_evidence != expected_request:
|
|
549
|
+
raise ValueError("request evidence differs from the exact wire request")
|
|
550
|
+
request_record = request_evidence
|
|
551
|
+
typed_output = BaseModel.model_dump(
|
|
552
|
+
response.value,
|
|
553
|
+
mode="json",
|
|
554
|
+
by_alias=False,
|
|
555
|
+
exclude_unset=False,
|
|
556
|
+
exclude_defaults=False,
|
|
557
|
+
exclude_none=False,
|
|
558
|
+
exclude_computed_fields=True,
|
|
559
|
+
round_trip=True,
|
|
560
|
+
warnings="error",
|
|
561
|
+
fallback=None,
|
|
562
|
+
serialize_as_any=False,
|
|
563
|
+
)
|
|
564
|
+
output_bytes = _canonical_evidence_bytes(typed_output)
|
|
565
|
+
if len(output_bytes) > MAX_STRUCTURED_OUTPUT_EVIDENCE_UTF8_BYTES:
|
|
566
|
+
raise ValueError("typed output exceeds the evidence bound")
|
|
567
|
+
record: dict[str, object] = {
|
|
568
|
+
"schema_version": STRUCTURED_OUTPUT_EVIDENCE_SCHEMA_VERSION,
|
|
569
|
+
"call_id": request.call_id.value,
|
|
570
|
+
"operation": request.operation,
|
|
571
|
+
"provider_response_id": response.provider_response_id,
|
|
572
|
+
"request_evidence_sha256": request_record["request_evidence_sha256"],
|
|
573
|
+
"output_tool_name": request.output_tool_name,
|
|
574
|
+
"output_schema_sha256": request_record["output_schema_sha256"],
|
|
575
|
+
"typed_output": typed_output,
|
|
576
|
+
"typed_output_sha256": hashlib.sha256(output_bytes).hexdigest(),
|
|
577
|
+
"typed_output_utf8_bytes": len(output_bytes),
|
|
578
|
+
}
|
|
579
|
+
record["output_evidence_sha256"] = hashlib.sha256(
|
|
580
|
+
_STRUCTURED_OUTPUT_EVIDENCE_DOMAIN + _canonical_evidence_bytes(record)
|
|
581
|
+
).hexdigest()
|
|
582
|
+
return record
|
|
583
|
+
|
|
584
|
+
|
|
585
|
+
def _canonical_evidence_mapping(
|
|
586
|
+
record: Mapping[str, object],
|
|
587
|
+
*,
|
|
588
|
+
label: str,
|
|
589
|
+
) -> dict[str, object]:
|
|
590
|
+
if not isinstance(record, Mapping):
|
|
591
|
+
raise TypeError(f"{label} must be a mapping")
|
|
592
|
+
try:
|
|
593
|
+
encoded = _canonical_evidence_bytes(dict(record))
|
|
594
|
+
decoded = json.loads(encoded)
|
|
595
|
+
except (TypeError, ValueError) as exc:
|
|
596
|
+
raise ValueError(f"{label} must contain canonical JSON values") from exc
|
|
597
|
+
if type(decoded) is not dict:
|
|
598
|
+
raise ValueError(f"{label} must encode one exact object")
|
|
599
|
+
return decoded
|
|
600
|
+
|
|
601
|
+
|
|
602
|
+
def _validate_evidence_sha256(value: object, *, field_name: str) -> str:
|
|
603
|
+
if type(value) is not str or _LOWER_SHA256.fullmatch(value) is None:
|
|
604
|
+
raise ValueError(f"{field_name} must be a lowercase SHA-256 digest")
|
|
605
|
+
return value
|
|
606
|
+
|
|
607
|
+
|
|
608
|
+
def _validate_evidence_identity_fields(record: Mapping[str, object]) -> None:
|
|
609
|
+
call_id = record["call_id"]
|
|
610
|
+
if type(call_id) is not str:
|
|
611
|
+
raise ValueError("call_id must be an exact string")
|
|
612
|
+
try:
|
|
613
|
+
LLMCallId(call_id)
|
|
614
|
+
except (TypeError, ValueError) as exc:
|
|
615
|
+
raise ValueError("call_id is outside the generic LLM identity domain") from exc
|
|
616
|
+
operation = record["operation"]
|
|
617
|
+
if type(operation) is not str or _EVIDENCE_OPERATION.fullmatch(operation) is None:
|
|
618
|
+
raise ValueError("operation is outside the closed token grammar")
|
|
619
|
+
tool_name = record["output_tool_name"]
|
|
620
|
+
if type(tool_name) is not str or _EVIDENCE_TOOL.fullmatch(tool_name) is None:
|
|
621
|
+
raise ValueError("output_tool_name is outside the closed tool grammar")
|
|
622
|
+
|
|
623
|
+
|
|
624
|
+
def validate_structured_generation_request_evidence_record(
|
|
625
|
+
record: Mapping[str, object],
|
|
626
|
+
) -> dict[str, object]:
|
|
627
|
+
"""Strictly verify and detach one persisted prequeue request record."""
|
|
628
|
+
|
|
629
|
+
canonical = _canonical_evidence_mapping(
|
|
630
|
+
record,
|
|
631
|
+
label="structured request evidence",
|
|
632
|
+
)
|
|
633
|
+
if frozenset(canonical) != _STRUCTURED_REQUEST_EVIDENCE_FIELDS:
|
|
634
|
+
raise ValueError("structured request evidence has unexpected fields")
|
|
635
|
+
if (
|
|
636
|
+
type(canonical["schema_version"]) is not int
|
|
637
|
+
or canonical["schema_version"] != STRUCTURED_REQUEST_EVIDENCE_SCHEMA_VERSION
|
|
638
|
+
):
|
|
639
|
+
raise ValueError("unsupported structured request evidence schema version")
|
|
640
|
+
_validate_evidence_identity_fields(canonical)
|
|
641
|
+
|
|
642
|
+
prompt_sha256 = _validate_evidence_sha256(
|
|
643
|
+
canonical["prompt_sha256"],
|
|
644
|
+
field_name="prompt_sha256",
|
|
645
|
+
)
|
|
646
|
+
wire_prompt_sha256 = _validate_evidence_sha256(
|
|
647
|
+
canonical["wire_prompt_sha256"],
|
|
648
|
+
field_name="wire_prompt_sha256",
|
|
649
|
+
)
|
|
650
|
+
if prompt_sha256 != wire_prompt_sha256:
|
|
651
|
+
raise ValueError("prompt_sha256 must equal its wire compatibility alias")
|
|
652
|
+
prompt_utf8_bytes = canonical["prompt_utf8_bytes"]
|
|
653
|
+
if (
|
|
654
|
+
type(prompt_utf8_bytes) is not int
|
|
655
|
+
or not 1 <= prompt_utf8_bytes <= MAX_PROMPT_UTF8_BYTES
|
|
656
|
+
):
|
|
657
|
+
raise ValueError("prompt_utf8_bytes is outside the generic prompt bound")
|
|
658
|
+
|
|
659
|
+
lineage_values = (
|
|
660
|
+
canonical["semantic_prompt_sha256"],
|
|
661
|
+
canonical["prompt_renderer_id"],
|
|
662
|
+
canonical["prompt_renderer_revision"],
|
|
663
|
+
canonical["prompt_renderer_definition_sha256"],
|
|
664
|
+
)
|
|
665
|
+
if not all(value is None for value in lineage_values):
|
|
666
|
+
if any(value is None for value in lineage_values):
|
|
667
|
+
raise ValueError("prompt lineage fields must be all present or all absent")
|
|
668
|
+
lineage = StructuredPromptLineage(
|
|
669
|
+
semantic_prompt_sha256=cast(str, lineage_values[0]),
|
|
670
|
+
renderer_id=cast(str, lineage_values[1]),
|
|
671
|
+
renderer_revision=cast(str, lineage_values[2]),
|
|
672
|
+
renderer_definition_sha256=cast(str, lineage_values[3]),
|
|
673
|
+
)
|
|
674
|
+
if lineage.renderer_id == IDENTITY_PROMPT_RENDERER_ID and (
|
|
675
|
+
lineage.semantic_prompt_sha256 != wire_prompt_sha256
|
|
676
|
+
or lineage.renderer_revision != IDENTITY_PROMPT_RENDERER_REVISION
|
|
677
|
+
or lineage.renderer_definition_sha256
|
|
678
|
+
!= IDENTITY_PROMPT_RENDERER_DEFINITION_SHA256
|
|
679
|
+
):
|
|
680
|
+
raise ValueError("identity renderer lineage is inconsistent")
|
|
681
|
+
|
|
682
|
+
output_type = canonical["output_type"]
|
|
683
|
+
if type(output_type) is not dict or frozenset(output_type) != {
|
|
684
|
+
"module",
|
|
685
|
+
"qualname",
|
|
686
|
+
}:
|
|
687
|
+
raise ValueError("output_type must contain exact module and qualname fields")
|
|
688
|
+
if any(
|
|
689
|
+
type(output_type[name]) is not str or not output_type[name]
|
|
690
|
+
for name in ("module", "qualname")
|
|
691
|
+
):
|
|
692
|
+
raise ValueError("output_type identities must be non-empty exact strings")
|
|
693
|
+
|
|
694
|
+
output_schema = canonical["output_schema"]
|
|
695
|
+
if type(output_schema) is not dict:
|
|
696
|
+
raise ValueError("output_schema must be an exact object")
|
|
697
|
+
schema_bytes = _canonical_evidence_bytes(output_schema)
|
|
698
|
+
if len(schema_bytes) > MAX_STRUCTURED_OUTPUT_SCHEMA_UTF8_BYTES:
|
|
699
|
+
raise ValueError("output_schema exceeds the evidence bound")
|
|
700
|
+
schema_utf8_bytes = canonical["output_schema_utf8_bytes"]
|
|
701
|
+
if type(schema_utf8_bytes) is not int or schema_utf8_bytes != len(schema_bytes):
|
|
702
|
+
raise ValueError("output_schema_utf8_bytes does not authenticate the schema")
|
|
703
|
+
schema_sha256 = _validate_evidence_sha256(
|
|
704
|
+
canonical["output_schema_sha256"],
|
|
705
|
+
field_name="output_schema_sha256",
|
|
706
|
+
)
|
|
707
|
+
if schema_sha256 != hashlib.sha256(schema_bytes).hexdigest():
|
|
708
|
+
raise ValueError("output_schema_sha256 does not authenticate the schema")
|
|
709
|
+
|
|
710
|
+
max_output_tokens = canonical["max_output_tokens"]
|
|
711
|
+
if (
|
|
712
|
+
type(max_output_tokens) is not int
|
|
713
|
+
or not 1 <= max_output_tokens <= MAX_OUTPUT_TOKENS
|
|
714
|
+
):
|
|
715
|
+
raise ValueError("max_output_tokens is outside the generic port bound")
|
|
716
|
+
temperature_hex = canonical["temperature_hex"]
|
|
717
|
+
if temperature_hex is not None:
|
|
718
|
+
if type(temperature_hex) is not str:
|
|
719
|
+
raise ValueError("temperature_hex must be an exact string or None")
|
|
720
|
+
try:
|
|
721
|
+
temperature = float.fromhex(temperature_hex)
|
|
722
|
+
except ValueError as exc:
|
|
723
|
+
raise ValueError(
|
|
724
|
+
"temperature_hex is not a finite hexadecimal float"
|
|
725
|
+
) from exc
|
|
726
|
+
if (
|
|
727
|
+
not math.isfinite(temperature)
|
|
728
|
+
or not 0 <= temperature <= 2
|
|
729
|
+
or temperature.hex() != temperature_hex
|
|
730
|
+
):
|
|
731
|
+
raise ValueError("temperature_hex is outside the canonical range")
|
|
732
|
+
|
|
733
|
+
supplied_sha256 = _validate_evidence_sha256(
|
|
734
|
+
canonical["request_evidence_sha256"],
|
|
735
|
+
field_name="request_evidence_sha256",
|
|
736
|
+
)
|
|
737
|
+
authenticated = dict(canonical)
|
|
738
|
+
del authenticated["request_evidence_sha256"]
|
|
739
|
+
expected_sha256 = hashlib.sha256(
|
|
740
|
+
_STRUCTURED_REQUEST_EVIDENCE_DOMAIN + _canonical_evidence_bytes(authenticated)
|
|
741
|
+
).hexdigest()
|
|
742
|
+
if supplied_sha256 != expected_sha256:
|
|
743
|
+
raise ValueError("request_evidence_sha256 does not authenticate the record")
|
|
744
|
+
return canonical
|
|
745
|
+
|
|
746
|
+
|
|
747
|
+
def validate_structured_generation_output_evidence_record(
|
|
748
|
+
record: Mapping[str, object],
|
|
749
|
+
*,
|
|
750
|
+
request_evidence: Mapping[str, object] | None = None,
|
|
751
|
+
) -> dict[str, object]:
|
|
752
|
+
"""Strictly verify one typed-output record and its optional request join."""
|
|
753
|
+
|
|
754
|
+
canonical = _canonical_evidence_mapping(
|
|
755
|
+
record,
|
|
756
|
+
label="structured output evidence",
|
|
757
|
+
)
|
|
758
|
+
if frozenset(canonical) != _STRUCTURED_OUTPUT_EVIDENCE_FIELDS:
|
|
759
|
+
raise ValueError("structured output evidence has unexpected fields")
|
|
760
|
+
if (
|
|
761
|
+
type(canonical["schema_version"]) is not int
|
|
762
|
+
or canonical["schema_version"] != STRUCTURED_OUTPUT_EVIDENCE_SCHEMA_VERSION
|
|
763
|
+
):
|
|
764
|
+
raise ValueError("unsupported structured output evidence schema version")
|
|
765
|
+
_validate_evidence_identity_fields(canonical)
|
|
766
|
+
provider_response_id = canonical["provider_response_id"]
|
|
767
|
+
if provider_response_id is not None and (
|
|
768
|
+
type(provider_response_id) is not str or not provider_response_id
|
|
769
|
+
):
|
|
770
|
+
raise ValueError("provider_response_id must be non-empty or None")
|
|
771
|
+
for name in (
|
|
772
|
+
"request_evidence_sha256",
|
|
773
|
+
"output_schema_sha256",
|
|
774
|
+
"typed_output_sha256",
|
|
775
|
+
"output_evidence_sha256",
|
|
776
|
+
):
|
|
777
|
+
_validate_evidence_sha256(canonical[name], field_name=name)
|
|
778
|
+
|
|
779
|
+
typed_output = canonical["typed_output"]
|
|
780
|
+
if type(typed_output) is not dict:
|
|
781
|
+
raise ValueError("typed_output must be an exact JSON object")
|
|
782
|
+
output_bytes = _canonical_evidence_bytes(typed_output)
|
|
783
|
+
if len(output_bytes) > MAX_STRUCTURED_OUTPUT_EVIDENCE_UTF8_BYTES:
|
|
784
|
+
raise ValueError("typed_output exceeds the evidence bound")
|
|
785
|
+
output_utf8_bytes = canonical["typed_output_utf8_bytes"]
|
|
786
|
+
if type(output_utf8_bytes) is not int or output_utf8_bytes != len(output_bytes):
|
|
787
|
+
raise ValueError("typed_output_utf8_bytes does not authenticate the output")
|
|
788
|
+
if canonical["typed_output_sha256"] != hashlib.sha256(output_bytes).hexdigest():
|
|
789
|
+
raise ValueError("typed_output_sha256 does not authenticate the output")
|
|
790
|
+
|
|
791
|
+
supplied_sha256 = canonical["output_evidence_sha256"]
|
|
792
|
+
authenticated = dict(canonical)
|
|
793
|
+
del authenticated["output_evidence_sha256"]
|
|
794
|
+
expected_sha256 = hashlib.sha256(
|
|
795
|
+
_STRUCTURED_OUTPUT_EVIDENCE_DOMAIN + _canonical_evidence_bytes(authenticated)
|
|
796
|
+
).hexdigest()
|
|
797
|
+
if supplied_sha256 != expected_sha256:
|
|
798
|
+
raise ValueError("output_evidence_sha256 does not authenticate the record")
|
|
799
|
+
|
|
800
|
+
if request_evidence is not None:
|
|
801
|
+
request_record = validate_structured_generation_request_evidence_record(
|
|
802
|
+
request_evidence
|
|
803
|
+
)
|
|
804
|
+
joined_fields = (
|
|
805
|
+
("call_id", "call_id"),
|
|
806
|
+
("operation", "operation"),
|
|
807
|
+
("output_tool_name", "output_tool_name"),
|
|
808
|
+
("output_schema_sha256", "output_schema_sha256"),
|
|
809
|
+
("request_evidence_sha256", "request_evidence_sha256"),
|
|
810
|
+
)
|
|
811
|
+
if any(
|
|
812
|
+
canonical[output_name] != request_record[request_name]
|
|
813
|
+
for output_name, request_name in joined_fields
|
|
814
|
+
):
|
|
815
|
+
raise ValueError("output evidence does not join its request evidence")
|
|
816
|
+
return canonical
|
|
817
|
+
|
|
818
|
+
|
|
819
|
+
def structured_generation_outcome_record(
|
|
820
|
+
outcome: LLMTaskOutcome[StructuredGenerationResponse[Any]],
|
|
821
|
+
) -> dict[str, object]:
|
|
822
|
+
"""Project a terminal outcome to sanitized, JSON-compatible telemetry.
|
|
823
|
+
|
|
824
|
+
The projection deliberately excludes prompts and typed output content. A
|
|
825
|
+
successful row retains the provider identity, usage, exact reported cost,
|
|
826
|
+
latency, and response identifier that would otherwise be lost if a later
|
|
827
|
+
experiment gate rejects the response. Schema version 2 added bounded,
|
|
828
|
+
sanitized failure evidence to each attempt; schema version 3 added bounded
|
|
829
|
+
structured-output diagnostics. Schema version 4 added the closed request
|
|
830
|
+
variant and SHA-256 for each prepared provider attempt. Schema version 5
|
|
831
|
+
binds deterministic physical-attempt identity and an optional closed stream
|
|
832
|
+
timeout phase. Schema version 6 added a finite canonical provider-error
|
|
833
|
+
code and a domain-separated fingerprint of a value-free redacted HTTP
|
|
834
|
+
error envelope. Schema version 8 added bounded, privacy-safe exception
|
|
835
|
+
provenance for otherwise-unknown adapter failures. The successful response
|
|
836
|
+
projection is otherwise unchanged.
|
|
837
|
+
"""
|
|
838
|
+
|
|
839
|
+
if type(outcome) is not LLMTaskOutcome:
|
|
840
|
+
raise TypeError("outcome must be an exact LLMTaskOutcome")
|
|
841
|
+
LLMTaskOutcome.__post_init__(outcome)
|
|
842
|
+
|
|
843
|
+
attempts: list[dict[str, object]] = []
|
|
844
|
+
for attempt in outcome.telemetry.attempts:
|
|
845
|
+
classification = attempt.classification
|
|
846
|
+
failure = None if classification is None else classification.sanitized_failure
|
|
847
|
+
attempts.append(
|
|
848
|
+
{
|
|
849
|
+
"attempt_number": attempt.attempt_number,
|
|
850
|
+
"status": attempt.status.value,
|
|
851
|
+
"wait_time_ns": attempt.wait_time_ns,
|
|
852
|
+
"service_time_ns": attempt.service_time_ns,
|
|
853
|
+
"will_retry": attempt.will_retry,
|
|
854
|
+
"policy_backoff_ns": attempt.policy_backoff_ns,
|
|
855
|
+
"retry_after_ns": attempt.retry_after_ns,
|
|
856
|
+
"scheduled_delay_ns": attempt.scheduled_delay_ns,
|
|
857
|
+
"error_type": attempt.error_type,
|
|
858
|
+
"request_evidence": (
|
|
859
|
+
None
|
|
860
|
+
if attempt.request_evidence is None
|
|
861
|
+
else {
|
|
862
|
+
"variant": attempt.request_evidence.variant.value,
|
|
863
|
+
"prompt_sha256": attempt.request_evidence.prompt_sha256,
|
|
864
|
+
"provider_attempt_id": (
|
|
865
|
+
None
|
|
866
|
+
if attempt.request_evidence.provider_attempt_id is None
|
|
867
|
+
else attempt.request_evidence.provider_attempt_id.value
|
|
868
|
+
),
|
|
869
|
+
}
|
|
870
|
+
),
|
|
871
|
+
"classification": (
|
|
872
|
+
None
|
|
873
|
+
if classification is None
|
|
874
|
+
else {
|
|
875
|
+
"disposition": classification.disposition.value,
|
|
876
|
+
"reason": classification.reason.value,
|
|
877
|
+
}
|
|
878
|
+
),
|
|
879
|
+
"failure": (
|
|
880
|
+
None
|
|
881
|
+
if failure is None
|
|
882
|
+
else {
|
|
883
|
+
"kind": failure.kind,
|
|
884
|
+
"retryable": failure.retryable,
|
|
885
|
+
"safe_message": failure.safe_message,
|
|
886
|
+
"status_code": failure.status_code,
|
|
887
|
+
"retry_after_seconds": failure.retry_after_seconds,
|
|
888
|
+
"provider_error_code": (
|
|
889
|
+
None
|
|
890
|
+
if failure.provider_error_code is None
|
|
891
|
+
else failure.provider_error_code.value
|
|
892
|
+
),
|
|
893
|
+
"provider_error_envelope_sha256": (
|
|
894
|
+
failure.provider_error_envelope_sha256
|
|
895
|
+
),
|
|
896
|
+
"exception_provenance": (
|
|
897
|
+
None
|
|
898
|
+
if failure.exception_provenance is None
|
|
899
|
+
else {
|
|
900
|
+
"truncated": (
|
|
901
|
+
failure.exception_provenance.truncated
|
|
902
|
+
),
|
|
903
|
+
"nodes": [
|
|
904
|
+
{
|
|
905
|
+
"parent_index": node.parent_index,
|
|
906
|
+
"link": node.link.value,
|
|
907
|
+
"family": node.family.value,
|
|
908
|
+
"type_identity_sha256": (
|
|
909
|
+
node.type_identity_sha256
|
|
910
|
+
),
|
|
911
|
+
}
|
|
912
|
+
for node in failure.exception_provenance.nodes
|
|
913
|
+
],
|
|
914
|
+
}
|
|
915
|
+
),
|
|
916
|
+
"stream_timeout_phase": (
|
|
917
|
+
None
|
|
918
|
+
if failure.stream_timeout_phase is None
|
|
919
|
+
else failure.stream_timeout_phase.value
|
|
920
|
+
),
|
|
921
|
+
"output_failure_mode": (
|
|
922
|
+
None
|
|
923
|
+
if failure.output_failure_mode is None
|
|
924
|
+
else failure.output_failure_mode.value
|
|
925
|
+
),
|
|
926
|
+
"validation_issues": [
|
|
927
|
+
{
|
|
928
|
+
"category": issue.category.value,
|
|
929
|
+
"location": list(issue.location),
|
|
930
|
+
"reason_code": (
|
|
931
|
+
None
|
|
932
|
+
if issue.reason_code is None
|
|
933
|
+
else issue.reason_code.value
|
|
934
|
+
),
|
|
935
|
+
}
|
|
936
|
+
for issue in failure.validation_issues
|
|
937
|
+
],
|
|
938
|
+
}
|
|
939
|
+
),
|
|
940
|
+
}
|
|
941
|
+
)
|
|
942
|
+
|
|
943
|
+
response_record: dict[str, object] | None = None
|
|
944
|
+
if outcome.status is TaskOutcomeStatus.SUCCEEDED:
|
|
945
|
+
response = outcome.response
|
|
946
|
+
if type(response) is not StructuredGenerationResponse:
|
|
947
|
+
raise TypeError("successful outcome has no structured response")
|
|
948
|
+
StructuredGenerationResponse.__post_init__(response)
|
|
949
|
+
response_record = {
|
|
950
|
+
"requested_model": response.requested_model,
|
|
951
|
+
"resolved_model": response.resolved_model,
|
|
952
|
+
"resolved_provider": response.resolved_provider,
|
|
953
|
+
"provider_response_id": response.provider_response_id,
|
|
954
|
+
"finish_reason": response.finish_reason,
|
|
955
|
+
"input_tokens": response.input_tokens,
|
|
956
|
+
"output_tokens": response.output_tokens,
|
|
957
|
+
"reasoning_tokens": response.reasoning_tokens,
|
|
958
|
+
"cache_read_tokens": response.cache_read_tokens,
|
|
959
|
+
"cache_write_tokens": response.cache_write_tokens,
|
|
960
|
+
"cost_usd": (None if response.cost_usd is None else str(response.cost_usd)),
|
|
961
|
+
"latency_ns": response.latency_ns,
|
|
962
|
+
}
|
|
963
|
+
|
|
964
|
+
return {
|
|
965
|
+
"schema_version": STRUCTURED_GENERATION_OUTCOME_SCHEMA_VERSION,
|
|
966
|
+
"task_id": outcome.telemetry.task_id,
|
|
967
|
+
"status": outcome.status.value,
|
|
968
|
+
"cancellation_reason": (
|
|
969
|
+
None
|
|
970
|
+
if outcome.cancellation_reason is None
|
|
971
|
+
else outcome.cancellation_reason.value
|
|
972
|
+
),
|
|
973
|
+
"queue_time_ns": outcome.telemetry.queue_time_ns,
|
|
974
|
+
"service_time_ns": outcome.telemetry.service_time_ns,
|
|
975
|
+
"total_time_ns": outcome.telemetry.total_time_ns,
|
|
976
|
+
"attempts": attempts,
|
|
977
|
+
"response": response_record,
|
|
978
|
+
}
|
|
979
|
+
|
|
980
|
+
|
|
981
|
+
@runtime_checkable
|
|
982
|
+
class StructuredAttemptRequestPolicy(Protocol):
|
|
983
|
+
"""Derive one attempt request from bounded queue context."""
|
|
984
|
+
|
|
985
|
+
def request_for_attempt(
|
|
986
|
+
self,
|
|
987
|
+
request: StructuredGenerationRequest[OutputT],
|
|
988
|
+
*,
|
|
989
|
+
context: LLMAttemptContext,
|
|
990
|
+
) -> "PreparedStructuredAttemptRequest[OutputT]": ...
|
|
991
|
+
|
|
992
|
+
|
|
993
|
+
@dataclass(frozen=True, slots=True)
|
|
994
|
+
class PreparedStructuredAttemptRequest(Generic[OutputT]):
|
|
995
|
+
"""Exact structured request paired with evidence derived from its prompt."""
|
|
996
|
+
|
|
997
|
+
request: StructuredGenerationRequest[OutputT]
|
|
998
|
+
evidence: AttemptRequestEvidence
|
|
999
|
+
|
|
1000
|
+
def __post_init__(self) -> None:
|
|
1001
|
+
if type(self.request) is not StructuredGenerationRequest:
|
|
1002
|
+
raise TypeError("request must be an exact StructuredGenerationRequest")
|
|
1003
|
+
StructuredGenerationRequest.__post_init__(self.request)
|
|
1004
|
+
if type(self.evidence) is not AttemptRequestEvidence:
|
|
1005
|
+
raise TypeError("evidence must be an AttemptRequestEvidence")
|
|
1006
|
+
expected = hashlib.sha256(
|
|
1007
|
+
self.request.prompt.encode("utf-8", errors="strict")
|
|
1008
|
+
).hexdigest()
|
|
1009
|
+
if self.evidence.prompt_sha256 != expected:
|
|
1010
|
+
raise ValueError("request evidence does not match the exact prompt")
|
|
1011
|
+
if self.evidence.provider_attempt_id != self.request.provider_attempt_id:
|
|
1012
|
+
raise ValueError(
|
|
1013
|
+
"request evidence and prepared request attempt identities differ"
|
|
1014
|
+
)
|
|
1015
|
+
|
|
1016
|
+
|
|
1017
|
+
def _provider_attempt_id(
|
|
1018
|
+
*,
|
|
1019
|
+
context: LLMAttemptContext,
|
|
1020
|
+
prompt_sha256: str,
|
|
1021
|
+
) -> ProviderAttemptId:
|
|
1022
|
+
"""Derive a content-free stable identity for one physical queue attempt."""
|
|
1023
|
+
|
|
1024
|
+
fields = (
|
|
1025
|
+
context.task_id.encode("utf-8", errors="strict"),
|
|
1026
|
+
str(context.attempt_number).encode("ascii"),
|
|
1027
|
+
prompt_sha256.encode("ascii", errors="strict"),
|
|
1028
|
+
)
|
|
1029
|
+
digest = hashlib.sha256(_PROVIDER_ATTEMPT_ID_DOMAIN)
|
|
1030
|
+
for field in fields:
|
|
1031
|
+
digest.update(len(field).to_bytes(8, "big"))
|
|
1032
|
+
digest.update(field)
|
|
1033
|
+
return ProviderAttemptId(f"provider_attempt_{digest.hexdigest()}")
|
|
1034
|
+
|
|
1035
|
+
|
|
1036
|
+
class ExactPayloadAttemptPolicy:
|
|
1037
|
+
"""Replay the original structured request byte-for-byte on every attempt.
|
|
1038
|
+
|
|
1039
|
+
This policy is useful for controlled replicates and transport-only retries
|
|
1040
|
+
where changing the prompt after a provider or validation failure would
|
|
1041
|
+
change the treatment. Retry admission remains owned by the queue and its
|
|
1042
|
+
classifier; this policy only guarantees that every admitted attempt uses
|
|
1043
|
+
the original prompt, output type, tool contract, and generation settings.
|
|
1044
|
+
"""
|
|
1045
|
+
|
|
1046
|
+
def request_for_attempt(
|
|
1047
|
+
self,
|
|
1048
|
+
request: StructuredGenerationRequest[OutputT],
|
|
1049
|
+
*,
|
|
1050
|
+
context: LLMAttemptContext,
|
|
1051
|
+
) -> PreparedStructuredAttemptRequest[OutputT]:
|
|
1052
|
+
if type(request) is not StructuredGenerationRequest:
|
|
1053
|
+
raise TypeError("request must be an exact StructuredGenerationRequest")
|
|
1054
|
+
StructuredGenerationRequest.__post_init__(request)
|
|
1055
|
+
if type(context) is not LLMAttemptContext:
|
|
1056
|
+
raise TypeError("context must be an exact LLMAttemptContext")
|
|
1057
|
+
LLMAttemptContext.__post_init__(context)
|
|
1058
|
+
return PreparedStructuredAttemptRequest(
|
|
1059
|
+
request=request,
|
|
1060
|
+
evidence=AttemptRequestEvidence(
|
|
1061
|
+
variant=AttemptRequestVariant.ORIGINAL,
|
|
1062
|
+
prompt_sha256=hashlib.sha256(
|
|
1063
|
+
request.prompt.encode("utf-8", errors="strict")
|
|
1064
|
+
).hexdigest(),
|
|
1065
|
+
),
|
|
1066
|
+
)
|
|
1067
|
+
|
|
1068
|
+
|
|
1069
|
+
class _SchemaRequiredPathMapUnavailable(ValueError):
|
|
1070
|
+
"""The local schema cannot yield one bounded, complete required-path map."""
|
|
1071
|
+
|
|
1072
|
+
|
|
1073
|
+
def _local_schema_reference(
|
|
1074
|
+
root: dict[str, Any],
|
|
1075
|
+
reference: object,
|
|
1076
|
+
) -> dict[str, Any] | bool:
|
|
1077
|
+
if type(reference) is not str or not reference.startswith("#/"):
|
|
1078
|
+
raise _SchemaRequiredPathMapUnavailable(
|
|
1079
|
+
"schema-repair path maps permit only local references"
|
|
1080
|
+
)
|
|
1081
|
+
current: object = root
|
|
1082
|
+
for raw_token in reference[2:].split("/"):
|
|
1083
|
+
token = raw_token.replace("~1", "/").replace("~0", "~")
|
|
1084
|
+
if type(current) is not dict or token not in current:
|
|
1085
|
+
raise _SchemaRequiredPathMapUnavailable(
|
|
1086
|
+
"schema-repair path map contains an unresolved reference"
|
|
1087
|
+
)
|
|
1088
|
+
current = current[token]
|
|
1089
|
+
if type(current) not in {dict, bool}:
|
|
1090
|
+
raise _SchemaRequiredPathMapUnavailable(
|
|
1091
|
+
"schema-repair reference does not resolve to a schema"
|
|
1092
|
+
)
|
|
1093
|
+
return current
|
|
1094
|
+
|
|
1095
|
+
|
|
1096
|
+
def _json_pointer(path: tuple[str, ...]) -> str:
|
|
1097
|
+
return "/" + "/".join(token.replace("~", "~0").replace("/", "~1") for token in path)
|
|
1098
|
+
|
|
1099
|
+
|
|
1100
|
+
def _required_field_paths(output_type: type[Any]) -> tuple[str, ...]:
|
|
1101
|
+
"""Enumerate all reachable ``required`` properties without partial output.
|
|
1102
|
+
|
|
1103
|
+
The map is derived solely from the trusted local Pydantic output type. If a
|
|
1104
|
+
recursive, malformed, or over-large schema cannot be represented in the
|
|
1105
|
+
fixed repair budget, callers retain the original request instead of giving
|
|
1106
|
+
the model an incomplete and therefore misleading field list.
|
|
1107
|
+
"""
|
|
1108
|
+
|
|
1109
|
+
try:
|
|
1110
|
+
root = output_type.model_json_schema()
|
|
1111
|
+
except Exception as error:
|
|
1112
|
+
raise _SchemaRequiredPathMapUnavailable(
|
|
1113
|
+
"local output schema generation failed"
|
|
1114
|
+
) from error
|
|
1115
|
+
if type(root) is not dict:
|
|
1116
|
+
raise _SchemaRequiredPathMapUnavailable("local output schema must be an object")
|
|
1117
|
+
|
|
1118
|
+
required_paths: set[tuple[str, ...]] = set()
|
|
1119
|
+
visited_nodes = 0
|
|
1120
|
+
|
|
1121
|
+
def visit(
|
|
1122
|
+
schema: object,
|
|
1123
|
+
path: tuple[str, ...],
|
|
1124
|
+
active_references: tuple[str, ...] = (),
|
|
1125
|
+
) -> None:
|
|
1126
|
+
nonlocal visited_nodes
|
|
1127
|
+
if type(schema) is bool:
|
|
1128
|
+
return
|
|
1129
|
+
if type(schema) is not dict:
|
|
1130
|
+
raise _SchemaRequiredPathMapUnavailable(
|
|
1131
|
+
"local output schema contains a malformed child"
|
|
1132
|
+
)
|
|
1133
|
+
visited_nodes += 1
|
|
1134
|
+
if visited_nodes > MAX_SCHEMA_REPAIR_SCHEMA_NODES:
|
|
1135
|
+
raise _SchemaRequiredPathMapUnavailable(
|
|
1136
|
+
"local output schema exceeds the node bound"
|
|
1137
|
+
)
|
|
1138
|
+
|
|
1139
|
+
if "$ref" in schema:
|
|
1140
|
+
reference = schema["$ref"]
|
|
1141
|
+
if type(reference) is not str or reference in active_references:
|
|
1142
|
+
raise _SchemaRequiredPathMapUnavailable(
|
|
1143
|
+
"recursive or malformed local output reference"
|
|
1144
|
+
)
|
|
1145
|
+
visit(
|
|
1146
|
+
_local_schema_reference(root, reference),
|
|
1147
|
+
path,
|
|
1148
|
+
(*active_references, reference),
|
|
1149
|
+
)
|
|
1150
|
+
siblings = {key: value for key, value in schema.items() if key != "$ref"}
|
|
1151
|
+
if siblings:
|
|
1152
|
+
visit(siblings, path, active_references)
|
|
1153
|
+
return
|
|
1154
|
+
|
|
1155
|
+
properties = schema.get("properties", {})
|
|
1156
|
+
if type(properties) is not dict:
|
|
1157
|
+
raise _SchemaRequiredPathMapUnavailable(
|
|
1158
|
+
"local output object properties are malformed"
|
|
1159
|
+
)
|
|
1160
|
+
required = schema.get("required", [])
|
|
1161
|
+
if type(required) is not list or not all(
|
|
1162
|
+
type(name) is str for name in required
|
|
1163
|
+
):
|
|
1164
|
+
raise _SchemaRequiredPathMapUnavailable(
|
|
1165
|
+
"local output required fields are malformed"
|
|
1166
|
+
)
|
|
1167
|
+
for name in required:
|
|
1168
|
+
required_paths.add((*path, name))
|
|
1169
|
+
if len(required_paths) > MAX_SCHEMA_REPAIR_REQUIRED_PATHS:
|
|
1170
|
+
raise _SchemaRequiredPathMapUnavailable(
|
|
1171
|
+
"local output schema exceeds the required-path bound"
|
|
1172
|
+
)
|
|
1173
|
+
for name, child in properties.items():
|
|
1174
|
+
if type(name) is not str:
|
|
1175
|
+
raise _SchemaRequiredPathMapUnavailable(
|
|
1176
|
+
"local output property name is malformed"
|
|
1177
|
+
)
|
|
1178
|
+
visit(child, (*path, name), active_references)
|
|
1179
|
+
|
|
1180
|
+
items = schema.get("items")
|
|
1181
|
+
if type(items) is list:
|
|
1182
|
+
for index, child in enumerate(items):
|
|
1183
|
+
visit(child, (*path, str(index)), active_references)
|
|
1184
|
+
elif items is not None:
|
|
1185
|
+
visit(items, (*path, "*"), active_references)
|
|
1186
|
+
prefix_items = schema.get("prefixItems")
|
|
1187
|
+
if prefix_items is not None:
|
|
1188
|
+
if type(prefix_items) is not list:
|
|
1189
|
+
raise _SchemaRequiredPathMapUnavailable(
|
|
1190
|
+
"local output tuple items are malformed"
|
|
1191
|
+
)
|
|
1192
|
+
for index, child in enumerate(prefix_items):
|
|
1193
|
+
visit(child, (*path, str(index)), active_references)
|
|
1194
|
+
|
|
1195
|
+
for keyword in ("allOf", "anyOf", "oneOf"):
|
|
1196
|
+
branches = schema.get(keyword)
|
|
1197
|
+
if branches is None:
|
|
1198
|
+
continue
|
|
1199
|
+
if type(branches) is not list:
|
|
1200
|
+
raise _SchemaRequiredPathMapUnavailable(
|
|
1201
|
+
"local output composition is malformed"
|
|
1202
|
+
)
|
|
1203
|
+
for branch in branches:
|
|
1204
|
+
visit(branch, path, active_references)
|
|
1205
|
+
for keyword in ("if", "then", "else", "not"):
|
|
1206
|
+
branch = schema.get(keyword)
|
|
1207
|
+
if branch is not None:
|
|
1208
|
+
visit(branch, path, active_references)
|
|
1209
|
+
|
|
1210
|
+
dependent_schemas = schema.get("dependentSchemas")
|
|
1211
|
+
if dependent_schemas is not None:
|
|
1212
|
+
if type(dependent_schemas) is not dict:
|
|
1213
|
+
raise _SchemaRequiredPathMapUnavailable(
|
|
1214
|
+
"local output dependent schemas are malformed"
|
|
1215
|
+
)
|
|
1216
|
+
for branch in dependent_schemas.values():
|
|
1217
|
+
visit(branch, path, active_references)
|
|
1218
|
+
|
|
1219
|
+
for keyword in ("patternProperties",):
|
|
1220
|
+
dynamic_schemas = schema.get(keyword)
|
|
1221
|
+
if dynamic_schemas is None:
|
|
1222
|
+
continue
|
|
1223
|
+
if type(dynamic_schemas) is not dict:
|
|
1224
|
+
raise _SchemaRequiredPathMapUnavailable(
|
|
1225
|
+
"local output dynamic properties are malformed"
|
|
1226
|
+
)
|
|
1227
|
+
for child in dynamic_schemas.values():
|
|
1228
|
+
visit(child, (*path, "*"), active_references)
|
|
1229
|
+
for keyword in ("additionalProperties", "unevaluatedProperties"):
|
|
1230
|
+
child = schema.get(keyword)
|
|
1231
|
+
if type(child) is dict:
|
|
1232
|
+
visit(child, (*path, "*"), active_references)
|
|
1233
|
+
for keyword in ("contains", "unevaluatedItems"):
|
|
1234
|
+
child = schema.get(keyword)
|
|
1235
|
+
if child is not None:
|
|
1236
|
+
visit(child, (*path, "*"), active_references)
|
|
1237
|
+
|
|
1238
|
+
visit(root, ())
|
|
1239
|
+
return tuple(sorted(_json_pointer(path) for path in required_paths))
|
|
1240
|
+
|
|
1241
|
+
|
|
1242
|
+
def _schema_repair_prompt_lineage(
|
|
1243
|
+
request: StructuredGenerationRequest[Any],
|
|
1244
|
+
) -> StructuredPromptLineage:
|
|
1245
|
+
upstream = request.prompt_lineage or identity_prompt_lineage(request.prompt)
|
|
1246
|
+
definition_record = {
|
|
1247
|
+
"schema_repair_policy_sha256": SCHEMA_REPAIR_POLICY_MANIFEST.policy_sha256,
|
|
1248
|
+
"upstream_renderer_id": upstream.renderer_id,
|
|
1249
|
+
"upstream_renderer_revision": upstream.renderer_revision,
|
|
1250
|
+
"upstream_renderer_definition_sha256": (upstream.renderer_definition_sha256),
|
|
1251
|
+
}
|
|
1252
|
+
definition_sha256 = hashlib.sha256(
|
|
1253
|
+
b"agent-evolve:schema-repair-prompt-renderer:v1\x00"
|
|
1254
|
+
+ _canonical_evidence_bytes(definition_record)
|
|
1255
|
+
).hexdigest()
|
|
1256
|
+
return StructuredPromptLineage(
|
|
1257
|
+
semantic_prompt_sha256=upstream.semantic_prompt_sha256,
|
|
1258
|
+
renderer_id=SCHEMA_REPAIR_PROMPT_RENDERER_ID,
|
|
1259
|
+
renderer_revision=SCHEMA_REPAIR_PROMPT_RENDERER_REVISION,
|
|
1260
|
+
renderer_definition_sha256=definition_sha256,
|
|
1261
|
+
)
|
|
1262
|
+
|
|
1263
|
+
|
|
1264
|
+
def _repair_literal_constraint_block(
|
|
1265
|
+
request: StructuredGenerationRequest[Any],
|
|
1266
|
+
failure: SanitizedAttemptFailure,
|
|
1267
|
+
) -> str:
|
|
1268
|
+
"""Render only trusted, provider-visible closed sets relevant to the failure."""
|
|
1269
|
+
|
|
1270
|
+
if not request.repair_literal_sets:
|
|
1271
|
+
return ""
|
|
1272
|
+
literal_failure = any(
|
|
1273
|
+
issue.category is ValidationIssueCategory.LITERAL_OR_ENUM
|
|
1274
|
+
or issue.reason_code
|
|
1275
|
+
is ValidationIssueReasonCode.FINITE_OPTION_OUT_OF_CONTRACT
|
|
1276
|
+
for issue in failure.validation_issues
|
|
1277
|
+
)
|
|
1278
|
+
if not literal_failure:
|
|
1279
|
+
return ""
|
|
1280
|
+
lines = [
|
|
1281
|
+
"Exact allowed string literals from the trusted local output contract "
|
|
1282
|
+
"(copy byte-for-byte; never synthesize or truncate an identifier):\n"
|
|
1283
|
+
]
|
|
1284
|
+
for constraint in request.repair_literal_sets:
|
|
1285
|
+
path = _json_pointer(constraint.field_path)
|
|
1286
|
+
literals = json.dumps(
|
|
1287
|
+
constraint.allowed_literals,
|
|
1288
|
+
ensure_ascii=True,
|
|
1289
|
+
separators=(",", ":"),
|
|
1290
|
+
)
|
|
1291
|
+
lines.append(f"- {path}={literals}\n")
|
|
1292
|
+
return "".join(lines)
|
|
1293
|
+
|
|
1294
|
+
|
|
1295
|
+
class SchemaRepairAttemptPolicy:
|
|
1296
|
+
"""Add bounded schema guidance only after a sanitized output failure."""
|
|
1297
|
+
|
|
1298
|
+
manifest = SCHEMA_REPAIR_POLICY_MANIFEST
|
|
1299
|
+
|
|
1300
|
+
@staticmethod
|
|
1301
|
+
def _location_text(location: tuple[str, ...]) -> str:
|
|
1302
|
+
return ".".join(location[:4])
|
|
1303
|
+
|
|
1304
|
+
@staticmethod
|
|
1305
|
+
def _prepared(
|
|
1306
|
+
request: StructuredGenerationRequest[OutputT],
|
|
1307
|
+
variant: AttemptRequestVariant,
|
|
1308
|
+
) -> PreparedStructuredAttemptRequest[OutputT]:
|
|
1309
|
+
evidence = AttemptRequestEvidence(
|
|
1310
|
+
variant=variant,
|
|
1311
|
+
prompt_sha256=hashlib.sha256(
|
|
1312
|
+
request.prompt.encode("utf-8", errors="strict")
|
|
1313
|
+
).hexdigest(),
|
|
1314
|
+
)
|
|
1315
|
+
return PreparedStructuredAttemptRequest(request=request, evidence=evidence)
|
|
1316
|
+
|
|
1317
|
+
def request_for_attempt(
|
|
1318
|
+
self,
|
|
1319
|
+
request: StructuredGenerationRequest[OutputT],
|
|
1320
|
+
*,
|
|
1321
|
+
context: LLMAttemptContext,
|
|
1322
|
+
) -> PreparedStructuredAttemptRequest[OutputT]:
|
|
1323
|
+
failure = context.active_output_failure
|
|
1324
|
+
if (
|
|
1325
|
+
failure is None
|
|
1326
|
+
or not failure.retryable
|
|
1327
|
+
or failure.kind != GenerationFailureKind.OUTPUT_INVALID.value
|
|
1328
|
+
):
|
|
1329
|
+
return self._prepared(request, AttemptRequestVariant.ORIGINAL)
|
|
1330
|
+
|
|
1331
|
+
mode = failure.output_failure_mode or (
|
|
1332
|
+
StructuredOutputFailureMode.TYPED_OUTPUT_CONTRACT
|
|
1333
|
+
)
|
|
1334
|
+
try:
|
|
1335
|
+
required_paths = _required_field_paths(request.output_type)
|
|
1336
|
+
except _SchemaRequiredPathMapUnavailable:
|
|
1337
|
+
return self._prepared(request, AttemptRequestVariant.ORIGINAL)
|
|
1338
|
+
required_paths_json = json.dumps(
|
|
1339
|
+
required_paths,
|
|
1340
|
+
ensure_ascii=True,
|
|
1341
|
+
separators=(",", ":"),
|
|
1342
|
+
)
|
|
1343
|
+
# Output-token pressure can surface as schema validation (for example,
|
|
1344
|
+
# a truncated object missing late fields), not only as the provider's
|
|
1345
|
+
# explicit incomplete-tool-call category. Keep this bounded guidance
|
|
1346
|
+
# active for every output-invalid repair without weakening the schema.
|
|
1347
|
+
completion_guidance = " Keep every field concise so the tool call completes."
|
|
1348
|
+
# A partitioned queue exposes semantic retry usage independently of
|
|
1349
|
+
# physical attempts. Deriving escalation from that ledger keeps a
|
|
1350
|
+
# repair request byte-identical across intervening 429/5xx/timeouts.
|
|
1351
|
+
# The legacy fallback preserves behavior for callers without the new
|
|
1352
|
+
# budget contract.
|
|
1353
|
+
if context.retry_budget_usage is not None:
|
|
1354
|
+
repair_pass = min(
|
|
1355
|
+
2,
|
|
1356
|
+
max(1, context.retry_budget_usage.output_invalid_retries),
|
|
1357
|
+
)
|
|
1358
|
+
else:
|
|
1359
|
+
repair_pass = (
|
|
1360
|
+
2
|
|
1361
|
+
if context.attempt_number >= 3
|
|
1362
|
+
and context.previous_failure is not None
|
|
1363
|
+
and context.previous_failure.kind
|
|
1364
|
+
== GenerationFailureKind.OUTPUT_INVALID.value
|
|
1365
|
+
else 1
|
|
1366
|
+
)
|
|
1367
|
+
escalation_guidance = (
|
|
1368
|
+
""
|
|
1369
|
+
if repair_pass == 1
|
|
1370
|
+
else (
|
|
1371
|
+
" FINAL BOUNDED REPAIR PASS: rebuild the complete tool call "
|
|
1372
|
+
"independently, then check every constrained string by exact "
|
|
1373
|
+
"equality against the trusted lists before emitting it."
|
|
1374
|
+
)
|
|
1375
|
+
)
|
|
1376
|
+
literal_constraint_block = _repair_literal_constraint_block(request, failure)
|
|
1377
|
+
|
|
1378
|
+
def render(issue_lines: list[str]) -> str:
|
|
1379
|
+
issue_block = (
|
|
1380
|
+
"Validation issues:\n" + "".join(issue_lines) if issue_lines else ""
|
|
1381
|
+
)
|
|
1382
|
+
return _SCHEMA_REPAIR_TEMPLATE.format(
|
|
1383
|
+
policy_version=SCHEMA_REPAIR_POLICY_VERSION,
|
|
1384
|
+
failure_mode=mode.value,
|
|
1385
|
+
repair_pass=repair_pass,
|
|
1386
|
+
required_paths_json=required_paths_json,
|
|
1387
|
+
issue_block=issue_block,
|
|
1388
|
+
literal_constraint_block=literal_constraint_block,
|
|
1389
|
+
output_tool_name=request.output_tool_name,
|
|
1390
|
+
completion_guidance=completion_guidance,
|
|
1391
|
+
escalation_guidance=escalation_guidance,
|
|
1392
|
+
)
|
|
1393
|
+
|
|
1394
|
+
if len(render([]).encode("utf-8", errors="strict")) > (
|
|
1395
|
+
MAX_SCHEMA_REPAIR_SUFFIX_UTF8_BYTES
|
|
1396
|
+
):
|
|
1397
|
+
return self._prepared(request, AttemptRequestVariant.ORIGINAL)
|
|
1398
|
+
issue_lines: list[str] = []
|
|
1399
|
+
for issue in failure.validation_issues:
|
|
1400
|
+
reason = (
|
|
1401
|
+
""
|
|
1402
|
+
if issue.reason_code is None
|
|
1403
|
+
else f"; reason={issue.reason_code.value}"
|
|
1404
|
+
)
|
|
1405
|
+
guidance = (
|
|
1406
|
+
""
|
|
1407
|
+
if issue.reason_code is None
|
|
1408
|
+
else (
|
|
1409
|
+
" "
|
|
1410
|
+
+ _SEMANTIC_REPAIR_GUIDANCE.get(
|
|
1411
|
+
issue.reason_code,
|
|
1412
|
+
_DEFAULT_SEMANTIC_REPAIR_GUIDANCE,
|
|
1413
|
+
)
|
|
1414
|
+
)
|
|
1415
|
+
)
|
|
1416
|
+
line = (
|
|
1417
|
+
f"- {issue.category.value} at "
|
|
1418
|
+
f"{self._location_text(issue.location)}{reason}."
|
|
1419
|
+
f"{guidance}\n"
|
|
1420
|
+
)
|
|
1421
|
+
candidate = render([*issue_lines, line])
|
|
1422
|
+
if (
|
|
1423
|
+
len(candidate.encode("utf-8", errors="strict"))
|
|
1424
|
+
> MAX_SCHEMA_REPAIR_SUFFIX_UTF8_BYTES
|
|
1425
|
+
):
|
|
1426
|
+
break
|
|
1427
|
+
issue_lines.append(line)
|
|
1428
|
+
suffix = render(issue_lines)
|
|
1429
|
+
if (
|
|
1430
|
+
len(suffix.encode("utf-8", errors="strict"))
|
|
1431
|
+
> MAX_SCHEMA_REPAIR_SUFFIX_UTF8_BYTES
|
|
1432
|
+
):
|
|
1433
|
+
raise AssertionError("schema repair suffix exceeded its static bound")
|
|
1434
|
+
repaired_prompt = request.prompt + suffix
|
|
1435
|
+
if (
|
|
1436
|
+
len(repaired_prompt.encode("utf-8", errors="strict"))
|
|
1437
|
+
> MAX_PROMPT_UTF8_BYTES
|
|
1438
|
+
):
|
|
1439
|
+
# A maximal original request remains a valid provider attempt. Do
|
|
1440
|
+
# not turn its retry into a local request-construction failure.
|
|
1441
|
+
return self._prepared(request, AttemptRequestVariant.ORIGINAL)
|
|
1442
|
+
repaired = replace(
|
|
1443
|
+
request,
|
|
1444
|
+
prompt=repaired_prompt,
|
|
1445
|
+
prompt_lineage=_schema_repair_prompt_lineage(request),
|
|
1446
|
+
)
|
|
1447
|
+
return self._prepared(repaired, AttemptRequestVariant.SCHEMA_REPAIR_V4)
|
|
1448
|
+
|
|
1449
|
+
|
|
1450
|
+
class ExactTransportSchemaRepairAttemptPolicy:
|
|
1451
|
+
"""Replay transport failures exactly and fail closed on repair derivation.
|
|
1452
|
+
|
|
1453
|
+
The ordinary :class:`SchemaRepairAttemptPolicy` deliberately falls back to
|
|
1454
|
+
the original request when it cannot derive a bounded, complete repair
|
|
1455
|
+
suffix. That is convenient in general-purpose applications, but it would
|
|
1456
|
+
turn a preregistered schema-repair attempt into an unlabelled additional
|
|
1457
|
+
sample. This experiment-facing policy therefore requires the authenticated
|
|
1458
|
+
repair variant whenever an output-invalid failure activated repair. All
|
|
1459
|
+
other admitted retries preserve the original request exactly.
|
|
1460
|
+
"""
|
|
1461
|
+
|
|
1462
|
+
manifest = SCHEMA_REPAIR_POLICY_MANIFEST
|
|
1463
|
+
|
|
1464
|
+
def __init__(self) -> None:
|
|
1465
|
+
self._exact = ExactPayloadAttemptPolicy()
|
|
1466
|
+
self._repair = SchemaRepairAttemptPolicy()
|
|
1467
|
+
|
|
1468
|
+
def request_for_attempt(
|
|
1469
|
+
self,
|
|
1470
|
+
request: StructuredGenerationRequest[OutputT],
|
|
1471
|
+
*,
|
|
1472
|
+
context: LLMAttemptContext,
|
|
1473
|
+
) -> PreparedStructuredAttemptRequest[OutputT]:
|
|
1474
|
+
if context.active_output_failure is None:
|
|
1475
|
+
return self._exact.request_for_attempt(request, context=context)
|
|
1476
|
+
prepared = self._repair.request_for_attempt(request, context=context)
|
|
1477
|
+
if prepared.evidence.variant is not AttemptRequestVariant.SCHEMA_REPAIR_V4:
|
|
1478
|
+
raise StructuredGenerationError(
|
|
1479
|
+
kind=GenerationFailureKind.INVALID_REQUEST,
|
|
1480
|
+
retryable=False,
|
|
1481
|
+
safe_message=(
|
|
1482
|
+
"bounded schema-repair guidance could not be derived locally"
|
|
1483
|
+
),
|
|
1484
|
+
)
|
|
1485
|
+
return prepared
|
|
1486
|
+
|
|
1487
|
+
|
|
1488
|
+
class StructuredGenerationExecutor:
|
|
1489
|
+
"""Execute exactly one structured-provider attempt for the queue."""
|
|
1490
|
+
|
|
1491
|
+
def __init__(
|
|
1492
|
+
self,
|
|
1493
|
+
generator: StructuredGenerator,
|
|
1494
|
+
*,
|
|
1495
|
+
attempt_request_policy: StructuredAttemptRequestPolicy | None = None,
|
|
1496
|
+
) -> None:
|
|
1497
|
+
if not isinstance(generator, StructuredGenerator):
|
|
1498
|
+
raise TypeError("generator must implement StructuredGenerator")
|
|
1499
|
+
if attempt_request_policy is None:
|
|
1500
|
+
attempt_request_policy = SchemaRepairAttemptPolicy()
|
|
1501
|
+
if not isinstance(attempt_request_policy, StructuredAttemptRequestPolicy):
|
|
1502
|
+
raise TypeError(
|
|
1503
|
+
"attempt_request_policy must implement StructuredAttemptRequestPolicy"
|
|
1504
|
+
)
|
|
1505
|
+
self.generator = generator
|
|
1506
|
+
self.attempt_request_policy = attempt_request_policy
|
|
1507
|
+
|
|
1508
|
+
def prepare_attempt(
|
|
1509
|
+
self,
|
|
1510
|
+
request: StructuredGenerationRequest[OutputT],
|
|
1511
|
+
*,
|
|
1512
|
+
context: LLMAttemptContext,
|
|
1513
|
+
) -> PreparedLLMAttempt[StructuredGenerationResponse[OutputT]]:
|
|
1514
|
+
if type(request) is not StructuredGenerationRequest:
|
|
1515
|
+
raise TypeError("request must be an exact StructuredGenerationRequest")
|
|
1516
|
+
StructuredGenerationRequest.__post_init__(request)
|
|
1517
|
+
if type(context) is not LLMAttemptContext:
|
|
1518
|
+
raise TypeError("context must be an exact LLMAttemptContext")
|
|
1519
|
+
LLMAttemptContext.__post_init__(context)
|
|
1520
|
+
|
|
1521
|
+
prepared_request = self.attempt_request_policy.request_for_attempt(
|
|
1522
|
+
request,
|
|
1523
|
+
context=context,
|
|
1524
|
+
)
|
|
1525
|
+
if type(prepared_request) is not PreparedStructuredAttemptRequest:
|
|
1526
|
+
raise TypeError("attempt request policy returned an invalid value")
|
|
1527
|
+
PreparedStructuredAttemptRequest.__post_init__(prepared_request)
|
|
1528
|
+
provider_attempt_id = _provider_attempt_id(
|
|
1529
|
+
context=context,
|
|
1530
|
+
prompt_sha256=prepared_request.evidence.prompt_sha256,
|
|
1531
|
+
)
|
|
1532
|
+
attempt_request = replace(
|
|
1533
|
+
prepared_request.request,
|
|
1534
|
+
provider_attempt_id=provider_attempt_id,
|
|
1535
|
+
)
|
|
1536
|
+
request_evidence = replace(
|
|
1537
|
+
prepared_request.evidence,
|
|
1538
|
+
provider_attempt_id=provider_attempt_id,
|
|
1539
|
+
)
|
|
1540
|
+
return PreparedLLMAttempt(
|
|
1541
|
+
execute_once=partial(
|
|
1542
|
+
self._execute_prepared,
|
|
1543
|
+
attempt_request,
|
|
1544
|
+
),
|
|
1545
|
+
request_evidence=request_evidence,
|
|
1546
|
+
)
|
|
1547
|
+
|
|
1548
|
+
async def execute(
|
|
1549
|
+
self,
|
|
1550
|
+
request: StructuredGenerationRequest[OutputT],
|
|
1551
|
+
*,
|
|
1552
|
+
context: LLMAttemptContext,
|
|
1553
|
+
) -> StructuredGenerationResponse[OutputT]:
|
|
1554
|
+
prepared = self.prepare_attempt(request, context=context)
|
|
1555
|
+
return await prepared.execute_once()
|
|
1556
|
+
|
|
1557
|
+
async def _execute_prepared(
|
|
1558
|
+
self,
|
|
1559
|
+
attempt_request: StructuredGenerationRequest[OutputT],
|
|
1560
|
+
) -> StructuredGenerationResponse[OutputT]:
|
|
1561
|
+
|
|
1562
|
+
# The queue still owns whether this attempt exists. The policy only
|
|
1563
|
+
# derives its request; the provider boundary never retries or sleeps.
|
|
1564
|
+
response = await self.generator.generate_once(attempt_request)
|
|
1565
|
+
if type(response) is not StructuredGenerationResponse:
|
|
1566
|
+
raise TypeError(
|
|
1567
|
+
"structured generator must return an exact StructuredGenerationResponse"
|
|
1568
|
+
)
|
|
1569
|
+
StructuredGenerationResponse.__post_init__(response)
|
|
1570
|
+
if type(response.value) is not attempt_request.output_type:
|
|
1571
|
+
raise TypeError("structured response value violates output_type")
|
|
1572
|
+
return response
|
|
1573
|
+
|
|
1574
|
+
|
|
1575
|
+
def _retry_after(seconds: float | None) -> RetryAfter | None:
|
|
1576
|
+
if seconds is None:
|
|
1577
|
+
return None
|
|
1578
|
+
# StructuredGenerationError already establishes finite, non-negative input.
|
|
1579
|
+
# Decimal(str(...)) plus ceiling prevents a positive sub-nanosecond server
|
|
1580
|
+
# delay from being shortened to zero.
|
|
1581
|
+
nanoseconds = int(
|
|
1582
|
+
(Decimal(str(seconds)) * NANOSECONDS_PER_SECOND).to_integral_value(
|
|
1583
|
+
rounding=ROUND_CEILING
|
|
1584
|
+
)
|
|
1585
|
+
)
|
|
1586
|
+
return RetryAfter(
|
|
1587
|
+
delay_ns=min(nanoseconds, _MAX_RETRY_AFTER_NS),
|
|
1588
|
+
source=RetryAfterSource.DELAY_SECONDS,
|
|
1589
|
+
)
|
|
1590
|
+
|
|
1591
|
+
|
|
1592
|
+
class StructuredGenerationRetryClassifier:
|
|
1593
|
+
"""Translate sanitized structured failures into the queue's closed domain."""
|
|
1594
|
+
|
|
1595
|
+
def classify(
|
|
1596
|
+
self,
|
|
1597
|
+
error: Exception,
|
|
1598
|
+
*,
|
|
1599
|
+
context: LLMAttemptContext,
|
|
1600
|
+
) -> RetryClassification:
|
|
1601
|
+
if type(context) is not LLMAttemptContext:
|
|
1602
|
+
raise TypeError("context must be an exact LLMAttemptContext")
|
|
1603
|
+
LLMAttemptContext.__post_init__(context)
|
|
1604
|
+
|
|
1605
|
+
if isinstance(error, TransportAbortedTimeoutError):
|
|
1606
|
+
return RetryClassification(
|
|
1607
|
+
disposition=RetryDisposition.FAIL,
|
|
1608
|
+
reason=RetryReason.TIMEOUT,
|
|
1609
|
+
sanitized_failure=SanitizedAttemptFailure(
|
|
1610
|
+
kind="timeout",
|
|
1611
|
+
retryable=False,
|
|
1612
|
+
safe_message=(
|
|
1613
|
+
"provider attempt exceeded its hard deadline; the owned "
|
|
1614
|
+
"transport was closed and the attempt was drained"
|
|
1615
|
+
),
|
|
1616
|
+
),
|
|
1617
|
+
)
|
|
1618
|
+
if not isinstance(error, StructuredGenerationError):
|
|
1619
|
+
if isinstance(error, TimeoutError):
|
|
1620
|
+
return RetryClassification(
|
|
1621
|
+
disposition=RetryDisposition.RETRY,
|
|
1622
|
+
reason=RetryReason.TIMEOUT,
|
|
1623
|
+
)
|
|
1624
|
+
return RetryClassification(
|
|
1625
|
+
disposition=RetryDisposition.FAIL,
|
|
1626
|
+
reason=RetryReason.INTERNAL,
|
|
1627
|
+
)
|
|
1628
|
+
|
|
1629
|
+
sanitized_failure = SanitizedAttemptFailure(
|
|
1630
|
+
kind=error.kind.value,
|
|
1631
|
+
retryable=error.retryable,
|
|
1632
|
+
safe_message=error.safe_message,
|
|
1633
|
+
status_code=error.status_code,
|
|
1634
|
+
retry_after_seconds=error.retry_after_seconds,
|
|
1635
|
+
output_failure_mode=error.output_failure_mode,
|
|
1636
|
+
validation_issues=error.validation_issues,
|
|
1637
|
+
provider_error_code=error.provider_error_code,
|
|
1638
|
+
provider_error_envelope_sha256=(error.provider_error_envelope_sha256),
|
|
1639
|
+
exception_provenance=error.exception_provenance,
|
|
1640
|
+
stream_timeout_phase=(
|
|
1641
|
+
error.phase
|
|
1642
|
+
if isinstance(
|
|
1643
|
+
error,
|
|
1644
|
+
(
|
|
1645
|
+
StructuredStreamTimeoutError,
|
|
1646
|
+
StructuredStreamCleanupTimeoutError,
|
|
1647
|
+
),
|
|
1648
|
+
)
|
|
1649
|
+
else None
|
|
1650
|
+
),
|
|
1651
|
+
)
|
|
1652
|
+
|
|
1653
|
+
disposition = (
|
|
1654
|
+
RetryDisposition.RETRY if error.retryable else RetryDisposition.FAIL
|
|
1655
|
+
)
|
|
1656
|
+
if error.kind is GenerationFailureKind.RATE_LIMITED:
|
|
1657
|
+
reason = RetryReason.RATE_LIMIT
|
|
1658
|
+
elif error.kind is GenerationFailureKind.TIMEOUT:
|
|
1659
|
+
reason = RetryReason.TIMEOUT
|
|
1660
|
+
elif error.kind is GenerationFailureKind.OUTPUT_INVALID:
|
|
1661
|
+
reason = RetryReason.OUTPUT_INVALID
|
|
1662
|
+
elif error.kind is GenerationFailureKind.PROVIDER_UNAVAILABLE:
|
|
1663
|
+
reason = RetryReason.TRANSIENT
|
|
1664
|
+
elif error.retryable:
|
|
1665
|
+
reason = RetryReason.TRANSIENT
|
|
1666
|
+
else:
|
|
1667
|
+
reason = RetryReason.PERMANENT
|
|
1668
|
+
|
|
1669
|
+
return RetryClassification(
|
|
1670
|
+
disposition=disposition,
|
|
1671
|
+
reason=reason,
|
|
1672
|
+
retry_after=(
|
|
1673
|
+
_retry_after(error.retry_after_seconds)
|
|
1674
|
+
if disposition is RetryDisposition.RETRY
|
|
1675
|
+
else None
|
|
1676
|
+
),
|
|
1677
|
+
sanitized_failure=sanitized_failure,
|
|
1678
|
+
)
|
|
1679
|
+
|
|
1680
|
+
|
|
1681
|
+
class TransportOnlyStructuredGenerationRetryClassifier:
|
|
1682
|
+
"""Retry transient transport conditions but never invalid model output.
|
|
1683
|
+
|
|
1684
|
+
The provider adapter may label incomplete or schema-invalid model output
|
|
1685
|
+
retryable for production repair workflows. Controlled experiments often
|
|
1686
|
+
need those failures to be terminal so that a physical retry cannot become
|
|
1687
|
+
an unplanned extra sample. HTTP status is authoritative: only 408, 429,
|
|
1688
|
+
and 500--599 may retry. Any other 4xx carrying a misleading transient kind
|
|
1689
|
+
or ``retryable=True`` remains terminal. Connection failures and
|
|
1690
|
+
cooperative stream-liveness timeouts have no HTTP status and retain the
|
|
1691
|
+
base classifier's retry behavior.
|
|
1692
|
+
"""
|
|
1693
|
+
|
|
1694
|
+
def __init__(self) -> None:
|
|
1695
|
+
self._base = StructuredGenerationRetryClassifier()
|
|
1696
|
+
|
|
1697
|
+
def classify(
|
|
1698
|
+
self,
|
|
1699
|
+
error: Exception,
|
|
1700
|
+
*,
|
|
1701
|
+
context: LLMAttemptContext,
|
|
1702
|
+
) -> RetryClassification:
|
|
1703
|
+
classified = self._base.classify(error, context=context)
|
|
1704
|
+
transport_condition = False
|
|
1705
|
+
if isinstance(error, StructuredGenerationError):
|
|
1706
|
+
if error.status_code is not None:
|
|
1707
|
+
transport_condition = (
|
|
1708
|
+
error.status_code in {408, 429} or 500 <= error.status_code <= 599
|
|
1709
|
+
)
|
|
1710
|
+
elif isinstance(error, StructuredStreamCleanupTimeoutError):
|
|
1711
|
+
transport_condition = False
|
|
1712
|
+
else:
|
|
1713
|
+
# Status-free TIMEOUT covers cooperative stream-liveness and
|
|
1714
|
+
# typed transport timeouts. Status-free PROVIDER_UNAVAILABLE
|
|
1715
|
+
# is the adapter's closed representation of a typed
|
|
1716
|
+
# connection failure. RATE_LIMITED is deliberately excluded:
|
|
1717
|
+
# the admitted representation of rate limiting is HTTP 429.
|
|
1718
|
+
transport_condition = error.kind in {
|
|
1719
|
+
GenerationFailureKind.TIMEOUT,
|
|
1720
|
+
GenerationFailureKind.PROVIDER_UNAVAILABLE,
|
|
1721
|
+
}
|
|
1722
|
+
ordinary_timeout = isinstance(error, TimeoutError) and not isinstance(
|
|
1723
|
+
error, TransportAbortedTimeoutError
|
|
1724
|
+
)
|
|
1725
|
+
if classified.disposition is RetryDisposition.RETRY and not (
|
|
1726
|
+
transport_condition or ordinary_timeout
|
|
1727
|
+
):
|
|
1728
|
+
return RetryClassification(
|
|
1729
|
+
disposition=RetryDisposition.FAIL,
|
|
1730
|
+
reason=classified.reason,
|
|
1731
|
+
sanitized_failure=classified.sanitized_failure,
|
|
1732
|
+
)
|
|
1733
|
+
return classified
|
|
1734
|
+
|
|
1735
|
+
|
|
1736
|
+
class NonRepeatingStreamTransportRetryClassifier:
|
|
1737
|
+
"""Retry transient pre-response transport failures, never an owned stream.
|
|
1738
|
+
|
|
1739
|
+
Once a streamed attempt has crossed the provider boundary, a first-event or
|
|
1740
|
+
idle-liveness timeout has an uncertain provider-side completion and billing
|
|
1741
|
+
state. Recovery/replay experiments therefore need a stricter policy than
|
|
1742
|
+
:class:`TransportOnlyStructuredGenerationRetryClassifier`: HTTP 408/429/5xx
|
|
1743
|
+
and typed connection failures may still retry, while every supervised
|
|
1744
|
+
stream timeout is terminal even when cancellation drained cleanly.
|
|
1745
|
+
"""
|
|
1746
|
+
|
|
1747
|
+
def __init__(self) -> None:
|
|
1748
|
+
self._transport_only = TransportOnlyStructuredGenerationRetryClassifier()
|
|
1749
|
+
|
|
1750
|
+
def classify(
|
|
1751
|
+
self,
|
|
1752
|
+
error: Exception,
|
|
1753
|
+
*,
|
|
1754
|
+
context: LLMAttemptContext,
|
|
1755
|
+
) -> RetryClassification:
|
|
1756
|
+
classified = self._transport_only.classify(error, context=context)
|
|
1757
|
+
if isinstance(error, StructuredStreamTimeoutError) and (
|
|
1758
|
+
classified.disposition is RetryDisposition.RETRY
|
|
1759
|
+
):
|
|
1760
|
+
return RetryClassification(
|
|
1761
|
+
disposition=RetryDisposition.FAIL,
|
|
1762
|
+
reason=classified.reason,
|
|
1763
|
+
sanitized_failure=classified.sanitized_failure,
|
|
1764
|
+
)
|
|
1765
|
+
return classified
|
|
1766
|
+
|
|
1767
|
+
|
|
1768
|
+
class OpaqueHTTP400OnceRetryClassifier:
|
|
1769
|
+
"""Retry one evidence-bearing but otherwise opaque HTTP 400 exactly once.
|
|
1770
|
+
|
|
1771
|
+
Some OpenRouter routes occasionally reject a byte-valid request before a
|
|
1772
|
+
stream exists while returning only an opaque HTTP-400 envelope. A later
|
|
1773
|
+
exact-payload replay can then succeed. This policy is deliberately much
|
|
1774
|
+
narrower than treating HTTP 400 as transient:
|
|
1775
|
+
|
|
1776
|
+
* only the first attempt is eligible;
|
|
1777
|
+
* the failure must be ``invalid_request`` with status 400;
|
|
1778
|
+
* a redacted envelope fingerprint must exist, while no typed provider code,
|
|
1779
|
+
output diagnostic, validation issue, or retry-after hint may exist; and
|
|
1780
|
+
* all ordinary non-repeating-stream transport rules remain unchanged.
|
|
1781
|
+
|
|
1782
|
+
Typed/actionable 4xx responses therefore remain terminal. Composition
|
|
1783
|
+
roots must also pair this classifier with an exact-payload attempt policy
|
|
1784
|
+
when request identity across the retry matters.
|
|
1785
|
+
"""
|
|
1786
|
+
|
|
1787
|
+
def __init__(self) -> None:
|
|
1788
|
+
self._non_repeating = NonRepeatingStreamTransportRetryClassifier()
|
|
1789
|
+
|
|
1790
|
+
def classify(
|
|
1791
|
+
self,
|
|
1792
|
+
error: Exception,
|
|
1793
|
+
*,
|
|
1794
|
+
context: LLMAttemptContext,
|
|
1795
|
+
) -> RetryClassification:
|
|
1796
|
+
classified = self._non_repeating.classify(error, context=context)
|
|
1797
|
+
if (
|
|
1798
|
+
classified.disposition is RetryDisposition.FAIL
|
|
1799
|
+
and context.attempt_number == 1
|
|
1800
|
+
and context.previous_failure is None
|
|
1801
|
+
and isinstance(error, StructuredGenerationError)
|
|
1802
|
+
and error.kind is GenerationFailureKind.INVALID_REQUEST
|
|
1803
|
+
and error.status_code == 400
|
|
1804
|
+
and error.provider_error_code is None
|
|
1805
|
+
and error.provider_error_envelope_sha256 is not None
|
|
1806
|
+
and error.retry_after_seconds is None
|
|
1807
|
+
and error.output_failure_mode is None
|
|
1808
|
+
and not error.validation_issues
|
|
1809
|
+
):
|
|
1810
|
+
return RetryClassification(
|
|
1811
|
+
disposition=RetryDisposition.RETRY,
|
|
1812
|
+
reason=RetryReason.TRANSIENT,
|
|
1813
|
+
sanitized_failure=classified.sanitized_failure,
|
|
1814
|
+
)
|
|
1815
|
+
return classified
|
|
1816
|
+
|
|
1817
|
+
|
|
1818
|
+
class BoundedOpaqueHTTP400RetryClassifier:
|
|
1819
|
+
"""Retry an identical opaque pre-stream HTTP 400 to the task budget.
|
|
1820
|
+
|
|
1821
|
+
A provider can transiently reject several byte-identical, contract-valid
|
|
1822
|
+
requests before accepting the next replay. This policy remains narrower
|
|
1823
|
+
than treating HTTP 400 as generally retryable:
|
|
1824
|
+
|
|
1825
|
+
* the response must be an ``invalid_request`` status 400 with a redacted
|
|
1826
|
+
envelope fingerprint and no typed provider code or validation detail;
|
|
1827
|
+
* every preceding failure in the replay chain must have the same envelope
|
|
1828
|
+
fingerprint and the same closed failure shape; and
|
|
1829
|
+
* the queue's immutable ``LLMTask.max_attempts`` remains the hard bound.
|
|
1830
|
+
|
|
1831
|
+
Composition roots must pair this classifier with an exact-payload attempt
|
|
1832
|
+
policy. Actionable 4xx responses, post-content stream failures, and a
|
|
1833
|
+
changed opaque envelope remain terminal.
|
|
1834
|
+
"""
|
|
1835
|
+
|
|
1836
|
+
def __init__(self) -> None:
|
|
1837
|
+
self._non_repeating = NonRepeatingStreamTransportRetryClassifier()
|
|
1838
|
+
|
|
1839
|
+
@staticmethod
|
|
1840
|
+
def _is_opaque_http_400(
|
|
1841
|
+
error: StructuredGenerationError,
|
|
1842
|
+
) -> bool:
|
|
1843
|
+
return (
|
|
1844
|
+
error.kind is GenerationFailureKind.INVALID_REQUEST
|
|
1845
|
+
and error.status_code == 400
|
|
1846
|
+
and error.provider_error_code is None
|
|
1847
|
+
and error.provider_error_envelope_sha256 is not None
|
|
1848
|
+
and error.retry_after_seconds is None
|
|
1849
|
+
and error.output_failure_mode is None
|
|
1850
|
+
and not error.validation_issues
|
|
1851
|
+
)
|
|
1852
|
+
|
|
1853
|
+
@staticmethod
|
|
1854
|
+
def _continues_same_chain(
|
|
1855
|
+
*,
|
|
1856
|
+
error: StructuredGenerationError,
|
|
1857
|
+
context: LLMAttemptContext,
|
|
1858
|
+
) -> bool:
|
|
1859
|
+
if context.attempt_number == 1:
|
|
1860
|
+
return context.previous_failure is None
|
|
1861
|
+
previous = context.previous_failure
|
|
1862
|
+
return (
|
|
1863
|
+
previous is not None
|
|
1864
|
+
and previous.kind
|
|
1865
|
+
== GenerationFailureKind.INVALID_REQUEST.value
|
|
1866
|
+
and previous.status_code == 400
|
|
1867
|
+
and previous.provider_error_code is None
|
|
1868
|
+
and previous.provider_error_envelope_sha256
|
|
1869
|
+
== error.provider_error_envelope_sha256
|
|
1870
|
+
and previous.retry_after_seconds is None
|
|
1871
|
+
and previous.output_failure_mode is None
|
|
1872
|
+
and not previous.validation_issues
|
|
1873
|
+
and context.active_output_failure is None
|
|
1874
|
+
)
|
|
1875
|
+
|
|
1876
|
+
def classify(
|
|
1877
|
+
self,
|
|
1878
|
+
error: Exception,
|
|
1879
|
+
*,
|
|
1880
|
+
context: LLMAttemptContext,
|
|
1881
|
+
) -> RetryClassification:
|
|
1882
|
+
classified = self._non_repeating.classify(
|
|
1883
|
+
error,
|
|
1884
|
+
context=context,
|
|
1885
|
+
)
|
|
1886
|
+
if not (
|
|
1887
|
+
classified.disposition is RetryDisposition.FAIL
|
|
1888
|
+
and isinstance(error, StructuredGenerationError)
|
|
1889
|
+
and self._is_opaque_http_400(error)
|
|
1890
|
+
and self._continues_same_chain(
|
|
1891
|
+
error=error,
|
|
1892
|
+
context=context,
|
|
1893
|
+
)
|
|
1894
|
+
):
|
|
1895
|
+
return classified
|
|
1896
|
+
return RetryClassification(
|
|
1897
|
+
disposition=RetryDisposition.RETRY,
|
|
1898
|
+
reason=RetryReason.TRANSIENT,
|
|
1899
|
+
sanitized_failure=classified.sanitized_failure,
|
|
1900
|
+
)
|
|
1901
|
+
|
|
1902
|
+
|
|
1903
|
+
class OpaqueHTTP400AndSchemaRepairOnceRetryClassifier:
|
|
1904
|
+
"""Combine exact opaque-400 recovery with one strict output repair.
|
|
1905
|
+
|
|
1906
|
+
Transport behavior is inherited unchanged from
|
|
1907
|
+
:class:`OpaqueHTTP400OnceRetryClassifier`, including terminal owned-stream
|
|
1908
|
+
timeouts and terminal typed/actionable 4xx responses. A retryable typed
|
|
1909
|
+
output failure receives one repair opportunity only. The queue's
|
|
1910
|
+
``active_output_failure`` marker prevents a second invalid output from
|
|
1911
|
+
becoming another model sample.
|
|
1912
|
+
"""
|
|
1913
|
+
|
|
1914
|
+
def __init__(self) -> None:
|
|
1915
|
+
self._opaque_http_400 = OpaqueHTTP400OnceRetryClassifier()
|
|
1916
|
+
self._structured = StructuredGenerationRetryClassifier()
|
|
1917
|
+
|
|
1918
|
+
def classify(
|
|
1919
|
+
self,
|
|
1920
|
+
error: Exception,
|
|
1921
|
+
*,
|
|
1922
|
+
context: LLMAttemptContext,
|
|
1923
|
+
) -> RetryClassification:
|
|
1924
|
+
classified = self._opaque_http_400.classify(error, context=context)
|
|
1925
|
+
if not (
|
|
1926
|
+
isinstance(error, StructuredGenerationError)
|
|
1927
|
+
and error.kind is GenerationFailureKind.OUTPUT_INVALID
|
|
1928
|
+
and error.retryable
|
|
1929
|
+
and context.active_output_failure is None
|
|
1930
|
+
):
|
|
1931
|
+
return classified
|
|
1932
|
+
repair = self._structured.classify(error, context=context)
|
|
1933
|
+
if (
|
|
1934
|
+
repair.disposition is RetryDisposition.RETRY
|
|
1935
|
+
and repair.reason is RetryReason.OUTPUT_INVALID
|
|
1936
|
+
):
|
|
1937
|
+
return repair
|
|
1938
|
+
return classified
|
|
1939
|
+
|
|
1940
|
+
|
|
1941
|
+
class OpaqueHTTP400AndBoundedSchemaRepairRetryClassifier:
|
|
1942
|
+
"""Combine opaque-400 replay with repair resampling to the task budget.
|
|
1943
|
+
|
|
1944
|
+
A typed output failure does not expose a valid candidate and therefore is
|
|
1945
|
+
not an optimization sample. Retrying it cannot select among candidate
|
|
1946
|
+
outcomes. This classifier admits another schema-repair attempt whenever
|
|
1947
|
+
the failure remains typed and retryable; the queue's immutable
|
|
1948
|
+
``LLMTask.max_attempts`` is the sole hard bound. Every physical attempt,
|
|
1949
|
+
exact repair prompt, and terminal response remains separately recorded.
|
|
1950
|
+
|
|
1951
|
+
Transport behavior is inherited unchanged from
|
|
1952
|
+
:class:`OpaqueHTTP400OnceRetryClassifier`: an opaque pre-stream HTTP 400
|
|
1953
|
+
may replay once, owned-stream timeouts remain terminal, and actionable 4xx
|
|
1954
|
+
responses never retry.
|
|
1955
|
+
"""
|
|
1956
|
+
|
|
1957
|
+
def __init__(self) -> None:
|
|
1958
|
+
self._opaque_http_400 = OpaqueHTTP400OnceRetryClassifier()
|
|
1959
|
+
self._structured = StructuredGenerationRetryClassifier()
|
|
1960
|
+
|
|
1961
|
+
def classify(
|
|
1962
|
+
self,
|
|
1963
|
+
error: Exception,
|
|
1964
|
+
*,
|
|
1965
|
+
context: LLMAttemptContext,
|
|
1966
|
+
) -> RetryClassification:
|
|
1967
|
+
classified = self._opaque_http_400.classify(error, context=context)
|
|
1968
|
+
if not (
|
|
1969
|
+
isinstance(error, StructuredGenerationError)
|
|
1970
|
+
and error.kind is GenerationFailureKind.OUTPUT_INVALID
|
|
1971
|
+
and error.retryable
|
|
1972
|
+
):
|
|
1973
|
+
return classified
|
|
1974
|
+
repair = self._structured.classify(error, context=context)
|
|
1975
|
+
if (
|
|
1976
|
+
repair.disposition is RetryDisposition.RETRY
|
|
1977
|
+
and repair.reason is RetryReason.OUTPUT_INVALID
|
|
1978
|
+
):
|
|
1979
|
+
return repair
|
|
1980
|
+
return classified
|
|
1981
|
+
|
|
1982
|
+
|
|
1983
|
+
class FirstEventResilientBoundedSchemaRepairRetryClassifier:
|
|
1984
|
+
"""Recover a content-blind first-event timeout inside one logical sample.
|
|
1985
|
+
|
|
1986
|
+
This policy preserves the opaque-HTTP-400 and bounded schema-repair
|
|
1987
|
+
semantics of :class:`OpaqueHTTP400AndBoundedSchemaRepairRetryClassifier`.
|
|
1988
|
+
It additionally retries a supervised ``FIRST_EVENT`` timeout because no
|
|
1989
|
+
provider content was observed and therefore no candidate outcome can be
|
|
1990
|
+
selected or discarded. ``IDLE`` and ``ABSOLUTE`` timeouts remain
|
|
1991
|
+
terminal because they follow observable stream progress, as do cleanup
|
|
1992
|
+
timeouts whose underlying attempt may still be running.
|
|
1993
|
+
|
|
1994
|
+
The queue's immutable attempt and partitioned retry budgets remain the
|
|
1995
|
+
hard bounds. Composition roots must pair this classifier with an exact
|
|
1996
|
+
transport/schema-repair attempt policy so the retry continues the same
|
|
1997
|
+
recorded logical sample rather than silently changing its prompt.
|
|
1998
|
+
"""
|
|
1999
|
+
|
|
2000
|
+
def __init__(self) -> None:
|
|
2001
|
+
self._bounded_schema_repair = (
|
|
2002
|
+
OpaqueHTTP400AndBoundedSchemaRepairRetryClassifier()
|
|
2003
|
+
)
|
|
2004
|
+
self._structured = StructuredGenerationRetryClassifier()
|
|
2005
|
+
|
|
2006
|
+
def classify(
|
|
2007
|
+
self,
|
|
2008
|
+
error: Exception,
|
|
2009
|
+
*,
|
|
2010
|
+
context: LLMAttemptContext,
|
|
2011
|
+
) -> RetryClassification:
|
|
2012
|
+
classified = self._bounded_schema_repair.classify(error, context=context)
|
|
2013
|
+
if not (
|
|
2014
|
+
isinstance(error, StructuredStreamTimeoutError)
|
|
2015
|
+
and error.phase is StructuredStreamTimeoutPhase.FIRST_EVENT
|
|
2016
|
+
):
|
|
2017
|
+
return classified
|
|
2018
|
+
retry = self._structured.classify(error, context=context)
|
|
2019
|
+
if (
|
|
2020
|
+
retry.disposition is RetryDisposition.RETRY
|
|
2021
|
+
and retry.reason is RetryReason.TIMEOUT
|
|
2022
|
+
):
|
|
2023
|
+
return retry
|
|
2024
|
+
return classified
|
|
2025
|
+
|
|
2026
|
+
|
|
2027
|
+
class BoundedPrestreamAndSchemaRepairRetryClassifier:
|
|
2028
|
+
"""Bound opaque pre-stream recovery, schema repair, and first-event retry.
|
|
2029
|
+
|
|
2030
|
+
This is the long-running campaign policy. It preserves exact request
|
|
2031
|
+
bytes across opaque HTTP-400 and transport retries, admits bounded repair
|
|
2032
|
+
resampling only after typed invalid output, and retries only the
|
|
2033
|
+
content-blind first-event stream timeout. Idle, absolute, and cleanup
|
|
2034
|
+
timeouts remain terminal.
|
|
2035
|
+
"""
|
|
2036
|
+
|
|
2037
|
+
def __init__(self) -> None:
|
|
2038
|
+
self._opaque_http_400 = BoundedOpaqueHTTP400RetryClassifier()
|
|
2039
|
+
self._structured = StructuredGenerationRetryClassifier()
|
|
2040
|
+
|
|
2041
|
+
def classify(
|
|
2042
|
+
self,
|
|
2043
|
+
error: Exception,
|
|
2044
|
+
*,
|
|
2045
|
+
context: LLMAttemptContext,
|
|
2046
|
+
) -> RetryClassification:
|
|
2047
|
+
classified = self._opaque_http_400.classify(
|
|
2048
|
+
error,
|
|
2049
|
+
context=context,
|
|
2050
|
+
)
|
|
2051
|
+
if (
|
|
2052
|
+
isinstance(error, StructuredGenerationError)
|
|
2053
|
+
and error.kind is GenerationFailureKind.OUTPUT_INVALID
|
|
2054
|
+
and error.retryable
|
|
2055
|
+
):
|
|
2056
|
+
repair = self._structured.classify(
|
|
2057
|
+
error,
|
|
2058
|
+
context=context,
|
|
2059
|
+
)
|
|
2060
|
+
if (
|
|
2061
|
+
repair.disposition is RetryDisposition.RETRY
|
|
2062
|
+
and repair.reason is RetryReason.OUTPUT_INVALID
|
|
2063
|
+
):
|
|
2064
|
+
return repair
|
|
2065
|
+
if not (
|
|
2066
|
+
isinstance(error, StructuredStreamTimeoutError)
|
|
2067
|
+
and error.phase is StructuredStreamTimeoutPhase.FIRST_EVENT
|
|
2068
|
+
):
|
|
2069
|
+
return classified
|
|
2070
|
+
retry = self._structured.classify(error, context=context)
|
|
2071
|
+
if (
|
|
2072
|
+
retry.disposition is RetryDisposition.RETRY
|
|
2073
|
+
and retry.reason is RetryReason.TIMEOUT
|
|
2074
|
+
):
|
|
2075
|
+
return retry
|
|
2076
|
+
return classified
|
|
2077
|
+
|
|
2078
|
+
|
|
2079
|
+
class QueuedStructuredGenerationError(RuntimeError):
|
|
2080
|
+
"""Sanitized non-success queue outcome with complete scheduling telemetry."""
|
|
2081
|
+
|
|
2082
|
+
def __init__(
|
|
2083
|
+
self,
|
|
2084
|
+
outcome: LLMTaskOutcome[StructuredGenerationResponse[Any]],
|
|
2085
|
+
) -> None:
|
|
2086
|
+
if type(outcome) is not LLMTaskOutcome:
|
|
2087
|
+
raise TypeError("outcome must be an exact LLMTaskOutcome")
|
|
2088
|
+
LLMTaskOutcome.__post_init__(outcome)
|
|
2089
|
+
if outcome.status is TaskOutcomeStatus.SUCCEEDED:
|
|
2090
|
+
raise ValueError("a successful outcome is not a terminal error")
|
|
2091
|
+
messages = {
|
|
2092
|
+
TaskOutcomeStatus.TERMINAL_FAILURE: (
|
|
2093
|
+
"queued structured generation failed terminally"
|
|
2094
|
+
),
|
|
2095
|
+
TaskOutcomeStatus.ATTEMPTS_EXHAUSTED: (
|
|
2096
|
+
"queued structured generation exhausted its attempt budget"
|
|
2097
|
+
),
|
|
2098
|
+
TaskOutcomeStatus.CANCELLED: "queued structured generation was cancelled",
|
|
2099
|
+
}
|
|
2100
|
+
super().__init__(messages[outcome.status])
|
|
2101
|
+
self.outcome = outcome
|
|
2102
|
+
|
|
2103
|
+
@property
|
|
2104
|
+
def status(self) -> TaskOutcomeStatus:
|
|
2105
|
+
return self.outcome.status
|
|
2106
|
+
|
|
2107
|
+
@property
|
|
2108
|
+
def telemetry(self) -> TaskTelemetry:
|
|
2109
|
+
return self.outcome.telemetry
|
|
2110
|
+
|
|
2111
|
+
@property
|
|
2112
|
+
def generation_failure_disposition(self) -> GenerationFailureDisposition:
|
|
2113
|
+
attempts = self.outcome.telemetry.attempts
|
|
2114
|
+
if not attempts:
|
|
2115
|
+
return GenerationFailureDisposition.INFRASTRUCTURE_FAILURE
|
|
2116
|
+
classification = attempts[-1].classification
|
|
2117
|
+
failure = None if classification is None else classification.sanitized_failure
|
|
2118
|
+
if failure is not None and failure.kind in {
|
|
2119
|
+
"output_invalid",
|
|
2120
|
+
"content_rejected",
|
|
2121
|
+
}:
|
|
2122
|
+
return GenerationFailureDisposition.MODEL_OR_SCHEMA_FAILURE
|
|
2123
|
+
return GenerationFailureDisposition.INFRASTRUCTURE_FAILURE
|
|
2124
|
+
|
|
2125
|
+
|
|
2126
|
+
class OutcomePublicationError(RuntimeError):
|
|
2127
|
+
"""Sanitized failure of a required terminal-outcome publication sink."""
|
|
2128
|
+
|
|
2129
|
+
def __init__(
|
|
2130
|
+
self,
|
|
2131
|
+
outcome: LLMTaskOutcome[StructuredGenerationResponse[Any]],
|
|
2132
|
+
) -> None:
|
|
2133
|
+
if type(outcome) is not LLMTaskOutcome:
|
|
2134
|
+
raise TypeError("outcome must be an exact LLMTaskOutcome")
|
|
2135
|
+
LLMTaskOutcome.__post_init__(outcome)
|
|
2136
|
+
super().__init__("required queued outcome publication failed")
|
|
2137
|
+
self.outcome = outcome
|
|
2138
|
+
|
|
2139
|
+
@property
|
|
2140
|
+
def status(self) -> TaskOutcomeStatus:
|
|
2141
|
+
return self.outcome.status
|
|
2142
|
+
|
|
2143
|
+
@property
|
|
2144
|
+
def telemetry(self) -> TaskTelemetry:
|
|
2145
|
+
return self.outcome.telemetry
|
|
2146
|
+
|
|
2147
|
+
@property
|
|
2148
|
+
def generation_failure_disposition(self) -> GenerationFailureDisposition:
|
|
2149
|
+
return GenerationFailureDisposition.INFRASTRUCTURE_FAILURE
|
|
2150
|
+
|
|
2151
|
+
|
|
2152
|
+
class CancelledOutcomePublicationError(asyncio.CancelledError):
|
|
2153
|
+
"""Cancellation whose required terminal receipt could not be published.
|
|
2154
|
+
|
|
2155
|
+
A submitter cancellation remains cancellation even when a required recorder
|
|
2156
|
+
fails: there is no provider response that can safely be released and no
|
|
2157
|
+
retry that can repair the recorder. This typed, content-free cancellation
|
|
2158
|
+
surfaces that secondary failure without replacing caller cancellation with
|
|
2159
|
+
an ordinary exception.
|
|
2160
|
+
"""
|
|
2161
|
+
|
|
2162
|
+
def __init__(
|
|
2163
|
+
self,
|
|
2164
|
+
outcome: LLMTaskOutcome[StructuredGenerationResponse[Any]],
|
|
2165
|
+
) -> None:
|
|
2166
|
+
if type(outcome) is not LLMTaskOutcome:
|
|
2167
|
+
raise TypeError("outcome must be an exact LLMTaskOutcome")
|
|
2168
|
+
LLMTaskOutcome.__post_init__(outcome)
|
|
2169
|
+
super().__init__("required cancelled-outcome publication failed")
|
|
2170
|
+
self.outcome = outcome
|
|
2171
|
+
|
|
2172
|
+
@property
|
|
2173
|
+
def status(self) -> TaskOutcomeStatus:
|
|
2174
|
+
return self.outcome.status
|
|
2175
|
+
|
|
2176
|
+
@property
|
|
2177
|
+
def telemetry(self) -> TaskTelemetry:
|
|
2178
|
+
return self.outcome.telemetry
|
|
2179
|
+
|
|
2180
|
+
|
|
2181
|
+
class StructuredEvidencePublicationError(RuntimeError):
|
|
2182
|
+
"""Sanitized failure of a required request/output evidence sink."""
|
|
2183
|
+
|
|
2184
|
+
def __init__(
|
|
2185
|
+
self,
|
|
2186
|
+
*,
|
|
2187
|
+
stage: StructuredEvidencePublicationStage,
|
|
2188
|
+
call_id: str,
|
|
2189
|
+
outcome: LLMTaskOutcome[StructuredGenerationResponse[Any]] | None = None,
|
|
2190
|
+
) -> None:
|
|
2191
|
+
if type(stage) is not StructuredEvidencePublicationStage:
|
|
2192
|
+
raise TypeError("stage must be a StructuredEvidencePublicationStage")
|
|
2193
|
+
if type(call_id) is not str or not call_id:
|
|
2194
|
+
raise ValueError("call_id must be a non-empty exact string")
|
|
2195
|
+
if outcome is not None:
|
|
2196
|
+
if type(outcome) is not LLMTaskOutcome:
|
|
2197
|
+
raise TypeError("outcome must be an exact LLMTaskOutcome or None")
|
|
2198
|
+
LLMTaskOutcome.__post_init__(outcome)
|
|
2199
|
+
if outcome.telemetry.task_id != call_id:
|
|
2200
|
+
raise ValueError("evidence failure outcome has a foreign call ID")
|
|
2201
|
+
super().__init__(
|
|
2202
|
+
f"required structured {stage.value} evidence publication failed"
|
|
2203
|
+
)
|
|
2204
|
+
self.stage = stage
|
|
2205
|
+
self.call_id = call_id
|
|
2206
|
+
self.outcome = outcome
|
|
2207
|
+
|
|
2208
|
+
@property
|
|
2209
|
+
def generation_failure_disposition(self) -> GenerationFailureDisposition:
|
|
2210
|
+
return GenerationFailureDisposition.INFRASTRUCTURE_FAILURE
|
|
2211
|
+
|
|
2212
|
+
|
|
2213
|
+
class QueuedStructuredGenerationRunner:
|
|
2214
|
+
"""Callable multi-attempt runner consumed by ``PydanticAIAgenticGenerator``."""
|
|
2215
|
+
|
|
2216
|
+
def __init__(
|
|
2217
|
+
self,
|
|
2218
|
+
*,
|
|
2219
|
+
queue: AsyncLLMTaskQueue[
|
|
2220
|
+
StructuredGenerationRequest[Any],
|
|
2221
|
+
StructuredGenerationResponse[Any],
|
|
2222
|
+
],
|
|
2223
|
+
max_attempts: int,
|
|
2224
|
+
retry_budget: PartitionedRetryBudget | None = None,
|
|
2225
|
+
owned_generator: PydanticAIStructuredGenerator | None = None,
|
|
2226
|
+
outcome_sink: OutcomeSink | None = None,
|
|
2227
|
+
outcome_publication_policy: OutcomePublicationPolicy = (
|
|
2228
|
+
OutcomePublicationPolicy.BEST_EFFORT
|
|
2229
|
+
),
|
|
2230
|
+
request_evidence_sink: StructuredRequestEvidenceSink | None = None,
|
|
2231
|
+
output_evidence_sink: StructuredOutputEvidenceSink | None = None,
|
|
2232
|
+
evidence_publication_policy: StructuredEvidencePublicationPolicy = (
|
|
2233
|
+
StructuredEvidencePublicationPolicy.BEST_EFFORT
|
|
2234
|
+
),
|
|
2235
|
+
) -> None:
|
|
2236
|
+
if type(queue) is not AsyncLLMTaskQueue:
|
|
2237
|
+
raise TypeError("queue must be an exact AsyncLLMTaskQueue")
|
|
2238
|
+
if type(max_attempts) is not int or not 1 <= max_attempts <= MAX_ATTEMPTS:
|
|
2239
|
+
raise ValueError(f"max_attempts must lie in [1, {MAX_ATTEMPTS}]")
|
|
2240
|
+
if retry_budget is not None and type(retry_budget) is not PartitionedRetryBudget:
|
|
2241
|
+
raise TypeError(
|
|
2242
|
+
"retry_budget must be a PartitionedRetryBudget or None"
|
|
2243
|
+
)
|
|
2244
|
+
if retry_budget is not None:
|
|
2245
|
+
PartitionedRetryBudget.__post_init__(retry_budget)
|
|
2246
|
+
if owned_generator is not None and not isinstance(
|
|
2247
|
+
owned_generator, PydanticAIStructuredGenerator
|
|
2248
|
+
):
|
|
2249
|
+
raise TypeError(
|
|
2250
|
+
"owned_generator must be a PydanticAIStructuredGenerator or None"
|
|
2251
|
+
)
|
|
2252
|
+
if outcome_sink is not None and not callable(outcome_sink):
|
|
2253
|
+
raise TypeError("outcome_sink must be callable or None")
|
|
2254
|
+
if type(outcome_publication_policy) is not OutcomePublicationPolicy:
|
|
2255
|
+
raise TypeError(
|
|
2256
|
+
"outcome_publication_policy must be an OutcomePublicationPolicy"
|
|
2257
|
+
)
|
|
2258
|
+
if (
|
|
2259
|
+
outcome_publication_policy is OutcomePublicationPolicy.REQUIRED
|
|
2260
|
+
and outcome_sink is None
|
|
2261
|
+
):
|
|
2262
|
+
raise ValueError("required outcome publication needs an outcome_sink")
|
|
2263
|
+
for name, sink in (
|
|
2264
|
+
("request_evidence_sink", request_evidence_sink),
|
|
2265
|
+
("output_evidence_sink", output_evidence_sink),
|
|
2266
|
+
):
|
|
2267
|
+
if sink is not None and not callable(sink):
|
|
2268
|
+
raise TypeError(f"{name} must be callable or None")
|
|
2269
|
+
if type(evidence_publication_policy) is not StructuredEvidencePublicationPolicy:
|
|
2270
|
+
raise TypeError(
|
|
2271
|
+
"evidence_publication_policy must be a "
|
|
2272
|
+
"StructuredEvidencePublicationPolicy"
|
|
2273
|
+
)
|
|
2274
|
+
if evidence_publication_policy is StructuredEvidencePublicationPolicy.REQUIRED:
|
|
2275
|
+
if request_evidence_sink is None or output_evidence_sink is None:
|
|
2276
|
+
raise ValueError(
|
|
2277
|
+
"required structured evidence publication needs both sinks"
|
|
2278
|
+
)
|
|
2279
|
+
self._queue = queue
|
|
2280
|
+
self.max_attempts = max_attempts
|
|
2281
|
+
self.retry_budget = retry_budget
|
|
2282
|
+
self._owned_generator = owned_generator
|
|
2283
|
+
self._outcome_sink = outcome_sink
|
|
2284
|
+
self.outcome_publication_policy = outcome_publication_policy
|
|
2285
|
+
self._request_evidence_sink = request_evidence_sink
|
|
2286
|
+
self._output_evidence_sink = output_evidence_sink
|
|
2287
|
+
self.evidence_publication_policy = evidence_publication_policy
|
|
2288
|
+
self._close_lock = asyncio.Lock()
|
|
2289
|
+
self._closed = False
|
|
2290
|
+
|
|
2291
|
+
async def __call__(
|
|
2292
|
+
self,
|
|
2293
|
+
request: StructuredGenerationRequest[OutputT],
|
|
2294
|
+
) -> AttemptedStructuredGenerationResponse[OutputT]:
|
|
2295
|
+
return await self.generate(request)
|
|
2296
|
+
|
|
2297
|
+
async def generate(
|
|
2298
|
+
self,
|
|
2299
|
+
request: StructuredGenerationRequest[OutputT],
|
|
2300
|
+
) -> AttemptedStructuredGenerationResponse[OutputT]:
|
|
2301
|
+
if type(request) is not StructuredGenerationRequest:
|
|
2302
|
+
raise TypeError("request must be an exact StructuredGenerationRequest")
|
|
2303
|
+
StructuredGenerationRequest.__post_init__(request)
|
|
2304
|
+
request_evidence = self._publish_request_evidence(request)
|
|
2305
|
+
outcome = await self._queue.submit(
|
|
2306
|
+
LLMTask(
|
|
2307
|
+
task_id=request.call_id.value,
|
|
2308
|
+
request=request,
|
|
2309
|
+
max_attempts=self.max_attempts,
|
|
2310
|
+
retry_budget=self.retry_budget,
|
|
2311
|
+
),
|
|
2312
|
+
cancellation_outcome_sink=self._observe_cancelled_outcome,
|
|
2313
|
+
)
|
|
2314
|
+
if type(outcome) is not LLMTaskOutcome:
|
|
2315
|
+
raise TypeError("queue returned a non-outcome value")
|
|
2316
|
+
LLMTaskOutcome.__post_init__(outcome)
|
|
2317
|
+
self._observe_outcome(outcome)
|
|
2318
|
+
|
|
2319
|
+
if outcome.status is not TaskOutcomeStatus.SUCCEEDED:
|
|
2320
|
+
raise QueuedStructuredGenerationError(outcome) from None
|
|
2321
|
+
response = outcome.response
|
|
2322
|
+
if type(response) is not StructuredGenerationResponse:
|
|
2323
|
+
raise TypeError("successful queue outcome has no structured response")
|
|
2324
|
+
StructuredGenerationResponse.__post_init__(response)
|
|
2325
|
+
self._publish_output_evidence(
|
|
2326
|
+
request,
|
|
2327
|
+
outcome,
|
|
2328
|
+
request_evidence=request_evidence,
|
|
2329
|
+
)
|
|
2330
|
+
return AttemptedStructuredGenerationResponse(
|
|
2331
|
+
response=response,
|
|
2332
|
+
attempt_count=len(outcome.telemetry.attempts),
|
|
2333
|
+
)
|
|
2334
|
+
|
|
2335
|
+
def _observe_outcome(
|
|
2336
|
+
self,
|
|
2337
|
+
outcome: LLMTaskOutcome[StructuredGenerationResponse[Any]],
|
|
2338
|
+
) -> None:
|
|
2339
|
+
if self._outcome_sink is None:
|
|
2340
|
+
return
|
|
2341
|
+
try:
|
|
2342
|
+
self._outcome_sink(outcome)
|
|
2343
|
+
except Exception:
|
|
2344
|
+
if self.outcome_publication_policy is OutcomePublicationPolicy.REQUIRED:
|
|
2345
|
+
# The logical provider outcome is already terminal. Fail
|
|
2346
|
+
# closed without retrying or exposing the response downstream.
|
|
2347
|
+
raise OutcomePublicationError(outcome) from None
|
|
2348
|
+
# Best-effort publication preserves the historical behavior: a
|
|
2349
|
+
# recorder failure does not change an already-terminal outcome.
|
|
2350
|
+
|
|
2351
|
+
def _observe_cancelled_outcome(
|
|
2352
|
+
self,
|
|
2353
|
+
outcome: LLMTaskOutcome[StructuredGenerationResponse[Any]],
|
|
2354
|
+
) -> None:
|
|
2355
|
+
"""Publish a cancelled await's terminal receipt without changing its law."""
|
|
2356
|
+
|
|
2357
|
+
try:
|
|
2358
|
+
self._observe_outcome(outcome)
|
|
2359
|
+
except OutcomePublicationError as error:
|
|
2360
|
+
# Required publication failure is observable, but remains a
|
|
2361
|
+
# cancellation so concurrent-stage cleanup cannot misclassify the
|
|
2362
|
+
# cancelled sibling as the primary model/provider failure.
|
|
2363
|
+
raise CancelledOutcomePublicationError(error.outcome) from None
|
|
2364
|
+
|
|
2365
|
+
def _publish_request_evidence(
|
|
2366
|
+
self,
|
|
2367
|
+
request: StructuredGenerationRequest[Any],
|
|
2368
|
+
) -> dict[str, object] | None:
|
|
2369
|
+
sink = self._request_evidence_sink
|
|
2370
|
+
if sink is None:
|
|
2371
|
+
return None
|
|
2372
|
+
try:
|
|
2373
|
+
record = structured_generation_request_evidence_record(request)
|
|
2374
|
+
sink(record)
|
|
2375
|
+
return record
|
|
2376
|
+
except Exception:
|
|
2377
|
+
if (
|
|
2378
|
+
self.evidence_publication_policy
|
|
2379
|
+
is StructuredEvidencePublicationPolicy.REQUIRED
|
|
2380
|
+
):
|
|
2381
|
+
raise StructuredEvidencePublicationError(
|
|
2382
|
+
stage=StructuredEvidencePublicationStage.REQUEST,
|
|
2383
|
+
call_id=request.call_id.value,
|
|
2384
|
+
) from None
|
|
2385
|
+
return None
|
|
2386
|
+
|
|
2387
|
+
def _publish_output_evidence(
|
|
2388
|
+
self,
|
|
2389
|
+
request: StructuredGenerationRequest[Any],
|
|
2390
|
+
outcome: LLMTaskOutcome[StructuredGenerationResponse[Any]],
|
|
2391
|
+
*,
|
|
2392
|
+
request_evidence: dict[str, object] | None,
|
|
2393
|
+
) -> None:
|
|
2394
|
+
sink = self._output_evidence_sink
|
|
2395
|
+
if sink is None:
|
|
2396
|
+
return
|
|
2397
|
+
try:
|
|
2398
|
+
record = structured_generation_output_evidence_record(
|
|
2399
|
+
request,
|
|
2400
|
+
outcome,
|
|
2401
|
+
request_evidence=request_evidence,
|
|
2402
|
+
)
|
|
2403
|
+
sink(record)
|
|
2404
|
+
except Exception:
|
|
2405
|
+
if (
|
|
2406
|
+
self.evidence_publication_policy
|
|
2407
|
+
is StructuredEvidencePublicationPolicy.REQUIRED
|
|
2408
|
+
):
|
|
2409
|
+
raise StructuredEvidencePublicationError(
|
|
2410
|
+
stage=StructuredEvidencePublicationStage.OUTPUT,
|
|
2411
|
+
call_id=request.call_id.value,
|
|
2412
|
+
outcome=outcome,
|
|
2413
|
+
) from None
|
|
2414
|
+
|
|
2415
|
+
async def snapshot(self) -> QueueSnapshot:
|
|
2416
|
+
return await self._queue.snapshot()
|
|
2417
|
+
|
|
2418
|
+
async def aclose(self) -> None:
|
|
2419
|
+
async with self._close_lock:
|
|
2420
|
+
if self._closed:
|
|
2421
|
+
return
|
|
2422
|
+
try:
|
|
2423
|
+
await self._queue.aclose()
|
|
2424
|
+
finally:
|
|
2425
|
+
self._closed = True
|
|
2426
|
+
if self._owned_generator is not None:
|
|
2427
|
+
generator = self._owned_generator
|
|
2428
|
+
self._owned_generator = None
|
|
2429
|
+
await generator.aclose()
|
|
2430
|
+
|
|
2431
|
+
async def __aenter__(self) -> "QueuedStructuredGenerationRunner":
|
|
2432
|
+
if self._closed:
|
|
2433
|
+
raise LLMTaskQueueClosedError("the queued runner is closed")
|
|
2434
|
+
return self
|
|
2435
|
+
|
|
2436
|
+
async def __aexit__(self, *_: object) -> None:
|
|
2437
|
+
await self.aclose()
|
|
2438
|
+
|
|
2439
|
+
|
|
2440
|
+
def _composed_runner(
|
|
2441
|
+
*,
|
|
2442
|
+
generator: StructuredGenerator,
|
|
2443
|
+
max_in_flight: int,
|
|
2444
|
+
max_pending: int,
|
|
2445
|
+
max_attempts: int,
|
|
2446
|
+
retry_budget: PartitionedRetryBudget | None,
|
|
2447
|
+
attempt_timeout_ns: int | None,
|
|
2448
|
+
backoff_policy: BackoffPolicy,
|
|
2449
|
+
runtime: AsyncRuntime,
|
|
2450
|
+
owned_generator: PydanticAIStructuredGenerator | None,
|
|
2451
|
+
outcome_sink: OutcomeSink | None,
|
|
2452
|
+
outcome_publication_policy: OutcomePublicationPolicy,
|
|
2453
|
+
request_evidence_sink: StructuredRequestEvidenceSink | None,
|
|
2454
|
+
output_evidence_sink: StructuredOutputEvidenceSink | None,
|
|
2455
|
+
evidence_publication_policy: StructuredEvidencePublicationPolicy,
|
|
2456
|
+
attempt_request_policy: StructuredAttemptRequestPolicy | None,
|
|
2457
|
+
retry_classifier: RetryClassifier,
|
|
2458
|
+
) -> QueuedStructuredGenerationRunner:
|
|
2459
|
+
queue = AsyncLLMTaskQueue(
|
|
2460
|
+
executor=StructuredGenerationExecutor(
|
|
2461
|
+
generator,
|
|
2462
|
+
attempt_request_policy=attempt_request_policy,
|
|
2463
|
+
),
|
|
2464
|
+
retry_classifier=retry_classifier,
|
|
2465
|
+
backoff_policy=backoff_policy,
|
|
2466
|
+
clock=SystemClock(),
|
|
2467
|
+
max_in_flight=max_in_flight,
|
|
2468
|
+
max_pending=max_pending,
|
|
2469
|
+
attempt_timeout_ns=attempt_timeout_ns,
|
|
2470
|
+
runtime=runtime,
|
|
2471
|
+
)
|
|
2472
|
+
return QueuedStructuredGenerationRunner(
|
|
2473
|
+
queue=queue,
|
|
2474
|
+
max_attempts=max_attempts,
|
|
2475
|
+
retry_budget=retry_budget,
|
|
2476
|
+
owned_generator=owned_generator,
|
|
2477
|
+
outcome_sink=outcome_sink,
|
|
2478
|
+
outcome_publication_policy=outcome_publication_policy,
|
|
2479
|
+
request_evidence_sink=request_evidence_sink,
|
|
2480
|
+
output_evidence_sink=output_evidence_sink,
|
|
2481
|
+
evidence_publication_policy=evidence_publication_policy,
|
|
2482
|
+
)
|
|
2483
|
+
|
|
2484
|
+
|
|
2485
|
+
def create_production_queued_runner(
|
|
2486
|
+
*,
|
|
2487
|
+
generator: PydanticAIStructuredGenerator,
|
|
2488
|
+
max_in_flight: int = DEFAULT_MAX_IN_FLIGHT,
|
|
2489
|
+
max_pending: int = DEFAULT_MAX_PENDING,
|
|
2490
|
+
max_attempts: int = DEFAULT_MAX_ATTEMPTS,
|
|
2491
|
+
retry_budget: PartitionedRetryBudget | None = None,
|
|
2492
|
+
attempt_timeout_ns: int | None = DEFAULT_ATTEMPT_TIMEOUT_NS,
|
|
2493
|
+
base_backoff_ns: int = DEFAULT_BASE_BACKOFF_NS,
|
|
2494
|
+
max_backoff_ns: int = DEFAULT_MAX_BACKOFF_NS,
|
|
2495
|
+
rate_limit_backoff_floor_ns: int = 0,
|
|
2496
|
+
random_source: RandomRange | None = None,
|
|
2497
|
+
jitter_policy: JitterPolicy | None = None,
|
|
2498
|
+
close_generator: bool = True,
|
|
2499
|
+
outcome_sink: OutcomeSink | None = None,
|
|
2500
|
+
outcome_publication_policy: OutcomePublicationPolicy = (
|
|
2501
|
+
OutcomePublicationPolicy.BEST_EFFORT
|
|
2502
|
+
),
|
|
2503
|
+
request_evidence_sink: StructuredRequestEvidenceSink | None = None,
|
|
2504
|
+
output_evidence_sink: StructuredOutputEvidenceSink | None = None,
|
|
2505
|
+
evidence_publication_policy: StructuredEvidencePublicationPolicy = (
|
|
2506
|
+
StructuredEvidencePublicationPolicy.BEST_EFFORT
|
|
2507
|
+
),
|
|
2508
|
+
attempt_request_policy: StructuredAttemptRequestPolicy | None = None,
|
|
2509
|
+
retry_classifier: RetryClassifier | None = None,
|
|
2510
|
+
) -> QueuedStructuredGenerationRunner:
|
|
2511
|
+
"""Build the real queue runtime around an already configured generator.
|
|
2512
|
+
|
|
2513
|
+
The existing OpenRouter generator factory fixes SDK and Pydantic-AI retries
|
|
2514
|
+
at zero. By default this factory takes ownership of that generator; pass
|
|
2515
|
+
``close_generator=False`` only when its lifecycle is owned elsewhere.
|
|
2516
|
+
Experiments that require durable pre-validation telemetry should supply an
|
|
2517
|
+
fsync-on-return sink and select ``OutcomePublicationPolicy.REQUIRED``.
|
|
2518
|
+
``attempt_request_policy`` lets an experiment bind the exact immutable
|
|
2519
|
+
schema-repair policy whose manifest is frozen with its launch contract.
|
|
2520
|
+
|
|
2521
|
+
``attempt_timeout_ns`` is the queue's absolute containment boundary, not a
|
|
2522
|
+
provider read-idle timeout. Set it to ``None`` when the generator owns a
|
|
2523
|
+
:class:`StructuredStreamLivenessPolicy`; the content-blind first-event and
|
|
2524
|
+
idle watchdogs then supervise normal liveness without imposing a fixed
|
|
2525
|
+
total cutoff on a progressing stream. An optional absolute fail-safe
|
|
2526
|
+
remains available in that policy. In this mode request cancellation stays
|
|
2527
|
+
local to its stream; the shared owned generator is closed only after queue
|
|
2528
|
+
shutdown has drained all active calls.
|
|
2529
|
+
"""
|
|
2530
|
+
|
|
2531
|
+
if not isinstance(generator, PydanticAIStructuredGenerator):
|
|
2532
|
+
raise TypeError("generator must be a PydanticAIStructuredGenerator")
|
|
2533
|
+
if generator.stream_liveness_policy is not None and attempt_timeout_ns is not None:
|
|
2534
|
+
raise ValueError(
|
|
2535
|
+
"progress-aware generators require attempt_timeout_ns=None; configure "
|
|
2536
|
+
"an absolute fail-safe on StructuredStreamLivenessPolicy instead"
|
|
2537
|
+
)
|
|
2538
|
+
if type(close_generator) is not bool:
|
|
2539
|
+
raise TypeError("close_generator must be bool")
|
|
2540
|
+
if retry_budget is not None and type(retry_budget) is not PartitionedRetryBudget:
|
|
2541
|
+
raise TypeError("retry_budget must be a PartitionedRetryBudget or None")
|
|
2542
|
+
if retry_budget is not None:
|
|
2543
|
+
PartitionedRetryBudget.__post_init__(retry_budget)
|
|
2544
|
+
if random_source is not None and jitter_policy is not None:
|
|
2545
|
+
raise ValueError("random_source and jitter_policy are mutually exclusive")
|
|
2546
|
+
if retry_classifier is None:
|
|
2547
|
+
retry_classifier = StructuredGenerationRetryClassifier()
|
|
2548
|
+
elif not isinstance(retry_classifier, RetryClassifier):
|
|
2549
|
+
raise TypeError("retry_classifier must implement RetryClassifier or be None")
|
|
2550
|
+
if jitter_policy is None:
|
|
2551
|
+
if random_source is None:
|
|
2552
|
+
random_source = SystemRandom()
|
|
2553
|
+
jitter_policy = FullJitter(random_source)
|
|
2554
|
+
elif not isinstance(jitter_policy, JitterPolicy):
|
|
2555
|
+
raise TypeError("jitter_policy must implement JitterPolicy or be None")
|
|
2556
|
+
backoff = ExponentialBackoff(
|
|
2557
|
+
base_delay_ns=base_backoff_ns,
|
|
2558
|
+
max_delay_ns=max_backoff_ns,
|
|
2559
|
+
jitter=jitter_policy,
|
|
2560
|
+
rate_limit_floor_ns=rate_limit_backoff_floor_ns,
|
|
2561
|
+
)
|
|
2562
|
+
return _composed_runner(
|
|
2563
|
+
generator=generator,
|
|
2564
|
+
max_in_flight=max_in_flight,
|
|
2565
|
+
max_pending=max_pending,
|
|
2566
|
+
max_attempts=max_attempts,
|
|
2567
|
+
retry_budget=retry_budget,
|
|
2568
|
+
attempt_timeout_ns=attempt_timeout_ns,
|
|
2569
|
+
backoff_policy=backoff,
|
|
2570
|
+
runtime=AsyncioRuntime(
|
|
2571
|
+
timeout_abort=(
|
|
2572
|
+
generator.aclose
|
|
2573
|
+
if close_generator and attempt_timeout_ns is not None
|
|
2574
|
+
else None
|
|
2575
|
+
),
|
|
2576
|
+
),
|
|
2577
|
+
owned_generator=generator if close_generator else None,
|
|
2578
|
+
outcome_sink=outcome_sink,
|
|
2579
|
+
outcome_publication_policy=outcome_publication_policy,
|
|
2580
|
+
request_evidence_sink=request_evidence_sink,
|
|
2581
|
+
output_evidence_sink=output_evidence_sink,
|
|
2582
|
+
evidence_publication_policy=evidence_publication_policy,
|
|
2583
|
+
attempt_request_policy=attempt_request_policy,
|
|
2584
|
+
retry_classifier=retry_classifier,
|
|
2585
|
+
)
|
|
2586
|
+
|
|
2587
|
+
|
|
2588
|
+
__all__ = [
|
|
2589
|
+
"CancelledOutcomePublicationError",
|
|
2590
|
+
"ExactPayloadAttemptPolicy",
|
|
2591
|
+
"ExactTransportSchemaRepairAttemptPolicy",
|
|
2592
|
+
"MAX_SCHEMA_REPAIR_REQUIRED_PATHS",
|
|
2593
|
+
"MAX_SCHEMA_REPAIR_SCHEMA_NODES",
|
|
2594
|
+
"MAX_SCHEMA_REPAIR_SUFFIX_UTF8_BYTES",
|
|
2595
|
+
"MAX_STRUCTURED_OUTPUT_EVIDENCE_UTF8_BYTES",
|
|
2596
|
+
"MAX_STRUCTURED_OUTPUT_SCHEMA_UTF8_BYTES",
|
|
2597
|
+
"OutcomePublicationError",
|
|
2598
|
+
"OutcomePublicationPolicy",
|
|
2599
|
+
"QueuedStructuredGenerationError",
|
|
2600
|
+
"QueuedStructuredGenerationRunner",
|
|
2601
|
+
"OutcomeSink",
|
|
2602
|
+
"PreparedStructuredAttemptRequest",
|
|
2603
|
+
"SCHEMA_REPAIR_POLICY_ID",
|
|
2604
|
+
"SCHEMA_REPAIR_POLICY_MANIFEST",
|
|
2605
|
+
"SCHEMA_REPAIR_POLICY_VERSION",
|
|
2606
|
+
"SchemaRepairPolicyManifest",
|
|
2607
|
+
"SchemaRepairAttemptPolicy",
|
|
2608
|
+
"StructuredAttemptRequestPolicy",
|
|
2609
|
+
"StructuredEvidencePublicationError",
|
|
2610
|
+
"StructuredEvidencePublicationPolicy",
|
|
2611
|
+
"StructuredEvidencePublicationStage",
|
|
2612
|
+
"STRUCTURED_OUTPUT_EVIDENCE_SCHEMA_VERSION",
|
|
2613
|
+
"STRUCTURED_REQUEST_EVIDENCE_SCHEMA_VERSION",
|
|
2614
|
+
"STRUCTURED_GENERATION_OUTCOME_SCHEMA_VERSION",
|
|
2615
|
+
"SUPPORTED_STRUCTURED_GENERATION_OUTCOME_SCHEMA_VERSIONS",
|
|
2616
|
+
"StructuredGenerationExecutor",
|
|
2617
|
+
"StructuredGenerationRetryClassifier",
|
|
2618
|
+
"NonRepeatingStreamTransportRetryClassifier",
|
|
2619
|
+
"BoundedOpaqueHTTP400RetryClassifier",
|
|
2620
|
+
"BoundedPrestreamAndSchemaRepairRetryClassifier",
|
|
2621
|
+
"OpaqueHTTP400OnceRetryClassifier",
|
|
2622
|
+
"OpaqueHTTP400AndSchemaRepairOnceRetryClassifier",
|
|
2623
|
+
"OpaqueHTTP400AndBoundedSchemaRepairRetryClassifier",
|
|
2624
|
+
"FirstEventResilientBoundedSchemaRepairRetryClassifier",
|
|
2625
|
+
"TransportOnlyStructuredGenerationRetryClassifier",
|
|
2626
|
+
"create_production_queued_runner",
|
|
2627
|
+
"structured_generation_output_evidence_record",
|
|
2628
|
+
"structured_generation_outcome_record",
|
|
2629
|
+
"structured_generation_request_evidence_record",
|
|
2630
|
+
"validate_structured_generation_output_evidence_record",
|
|
2631
|
+
"validate_structured_generation_request_evidence_record",
|
|
2632
|
+
"StructuredOutputEvidenceSink",
|
|
2633
|
+
"StructuredRequestEvidenceSink",
|
|
2634
|
+
]
|