agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
fi/opt/observability.py
ADDED
|
@@ -0,0 +1,4639 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import copy
|
|
4
|
+
import hashlib
|
|
5
|
+
import json
|
|
6
|
+
import os
|
|
7
|
+
import re
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
from typing import Any, Mapping, Optional, Sequence
|
|
10
|
+
|
|
11
|
+
from pydantic import BaseModel, Field
|
|
12
|
+
|
|
13
|
+
from .targets import AgentCandidate, CandidateEvaluation
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
OBSERVABILITY_SCHEMA_VERSION = "agent-opt.observability.v1"
|
|
17
|
+
REGRESSION_DATASET_SCHEMA_VERSION = "agent-opt.regression-dataset.v1"
|
|
18
|
+
REGRESSION_DATASET_COVERAGE_SCHEMA_VERSION = "agent-opt.regression-dataset-coverage.v1"
|
|
19
|
+
DATASET_SINK_SCHEMA_VERSION = "agent-opt.futureagi-dataset-sink.v1"
|
|
20
|
+
FUTUREAGI_EXPERIMENT_HISTORY_SCHEMA_VERSION = "agent-opt.futureagi-experiment-history.v1"
|
|
21
|
+
REGISTRY_REPLAY_PACK_MANIFEST_SCHEMA_VERSION = "agent-opt.registry-replay-pack.v1"
|
|
22
|
+
REGISTRY_REPLAY_PACK_PROMOTION_SCHEMA_VERSION = "agent-opt.registry-replay-pack-promotion.v1"
|
|
23
|
+
REGISTRY_REPLAY_PACK_LINEAGE_SCHEMA_VERSION = "agent-opt.registry-replay-pack-lineage.v1"
|
|
24
|
+
REGISTRY_REPLAY_PACK_TRIAGE_SCHEMA_VERSION = "agent-opt.registry-replay-pack-triage.v1"
|
|
25
|
+
FUTUREAGI_REGRESSION_DATASET_COLUMNS = (
|
|
26
|
+
{"name": "case_id", "data_type": "text"},
|
|
27
|
+
{"name": "query", "data_type": "text"},
|
|
28
|
+
{"name": "response", "data_type": "text"},
|
|
29
|
+
{"name": "expected_response", "data_type": "json"},
|
|
30
|
+
{"name": "observability", "data_type": "json"},
|
|
31
|
+
{"name": "tags", "data_type": "array"},
|
|
32
|
+
{"name": "metadata", "data_type": "json"},
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
class AgentObservabilityRecord(BaseModel):
|
|
37
|
+
"""One normalized production trace/evaluation record."""
|
|
38
|
+
|
|
39
|
+
index: int
|
|
40
|
+
source: str
|
|
41
|
+
framework: str
|
|
42
|
+
run_id: Optional[str] = None
|
|
43
|
+
candidate_id: Optional[str] = None
|
|
44
|
+
score: float
|
|
45
|
+
passed: bool
|
|
46
|
+
failures: list[str] = Field(default_factory=list)
|
|
47
|
+
metrics: dict[str, float] = Field(default_factory=dict)
|
|
48
|
+
trace_signals: list[str] = Field(default_factory=list)
|
|
49
|
+
raw: dict[str, Any] = Field(default_factory=dict)
|
|
50
|
+
metadata: dict[str, Any] = Field(default_factory=dict)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
class AgentObservabilityWindow(BaseModel):
|
|
54
|
+
"""A live or exported observability window ready for rollback monitoring."""
|
|
55
|
+
|
|
56
|
+
schema_version: str = OBSERVABILITY_SCHEMA_VERSION
|
|
57
|
+
source: str
|
|
58
|
+
framework: str
|
|
59
|
+
candidate: Optional[AgentCandidate] = None
|
|
60
|
+
records: list[AgentObservabilityRecord] = Field(default_factory=list)
|
|
61
|
+
required_metrics: dict[str, float] = Field(default_factory=dict)
|
|
62
|
+
required_trace_signals: list[str] = Field(default_factory=list)
|
|
63
|
+
metadata: dict[str, Any] = Field(default_factory=dict)
|
|
64
|
+
|
|
65
|
+
@property
|
|
66
|
+
def failures(self) -> list[str]:
|
|
67
|
+
failures: list[str] = []
|
|
68
|
+
for record in self.records:
|
|
69
|
+
failures.extend(record.failures)
|
|
70
|
+
return failures
|
|
71
|
+
|
|
72
|
+
@property
|
|
73
|
+
def average_score(self) -> Optional[float]:
|
|
74
|
+
if not self.records:
|
|
75
|
+
return None
|
|
76
|
+
return sum(record.score for record in self.records) / len(self.records)
|
|
77
|
+
|
|
78
|
+
def to_live_evaluations(
|
|
79
|
+
self,
|
|
80
|
+
*,
|
|
81
|
+
candidate: Optional[AgentCandidate] = None,
|
|
82
|
+
) -> list[CandidateEvaluation]:
|
|
83
|
+
active_candidate = candidate or self.candidate
|
|
84
|
+
return [
|
|
85
|
+
_evaluation_from_observability_record(record, candidate=active_candidate)
|
|
86
|
+
for record in self.records
|
|
87
|
+
]
|
|
88
|
+
|
|
89
|
+
def to_manifest(self) -> dict[str, Any]:
|
|
90
|
+
return self.model_dump()
|
|
91
|
+
|
|
92
|
+
def to_json(self, *, indent: int = 2) -> str:
|
|
93
|
+
return json.dumps(self.to_manifest(), sort_keys=True, indent=indent, default=str)
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
class AgentRegressionCase(BaseModel):
|
|
97
|
+
"""One replayable regression case derived from production observability."""
|
|
98
|
+
|
|
99
|
+
id: str
|
|
100
|
+
input: dict[str, Any]
|
|
101
|
+
expected: dict[str, Any]
|
|
102
|
+
tags: list[str] = Field(default_factory=list)
|
|
103
|
+
metadata: dict[str, Any] = Field(default_factory=dict)
|
|
104
|
+
|
|
105
|
+
def to_record(self) -> dict[str, Any]:
|
|
106
|
+
return self.model_dump()
|
|
107
|
+
|
|
108
|
+
def to_futureagi_row(self) -> dict[str, Any]:
|
|
109
|
+
observability = copy.deepcopy(self.input.get("observability", self.input))
|
|
110
|
+
expected = copy.deepcopy(self.expected)
|
|
111
|
+
metadata = {
|
|
112
|
+
**copy.deepcopy(self.metadata),
|
|
113
|
+
"dataset_case_id": self.id,
|
|
114
|
+
"tags": list(self.tags),
|
|
115
|
+
}
|
|
116
|
+
run_id = observability.get("run_id") if isinstance(observability, Mapping) else None
|
|
117
|
+
source = observability.get("source") if isinstance(observability, Mapping) else None
|
|
118
|
+
framework = observability.get("framework") if isinstance(observability, Mapping) else None
|
|
119
|
+
failures = observability.get("failures", []) if isinstance(observability, Mapping) else []
|
|
120
|
+
query = "Replay observability regression case"
|
|
121
|
+
if run_id:
|
|
122
|
+
query += f" {run_id}"
|
|
123
|
+
if source or framework:
|
|
124
|
+
query += f" from {source or 'unknown'}/{framework or 'unknown'}"
|
|
125
|
+
response = "; ".join(str(item) for item in failures) if failures else "passed"
|
|
126
|
+
return {
|
|
127
|
+
"case_id": self.id,
|
|
128
|
+
"query": query,
|
|
129
|
+
"response": response,
|
|
130
|
+
"expected_response": expected,
|
|
131
|
+
"observability": observability,
|
|
132
|
+
"tags": list(self.tags),
|
|
133
|
+
"metadata": metadata,
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
class AgentRegressionDataset(BaseModel):
|
|
138
|
+
"""A durable regression/replay dataset built from observability windows."""
|
|
139
|
+
|
|
140
|
+
schema_version: str = REGRESSION_DATASET_SCHEMA_VERSION
|
|
141
|
+
name: str
|
|
142
|
+
source: str
|
|
143
|
+
framework: str
|
|
144
|
+
cases: list[AgentRegressionCase] = Field(default_factory=list)
|
|
145
|
+
metadata: dict[str, Any] = Field(default_factory=dict)
|
|
146
|
+
|
|
147
|
+
def to_records(self) -> list[dict[str, Any]]:
|
|
148
|
+
return [case.to_record() for case in self.cases]
|
|
149
|
+
|
|
150
|
+
def to_jsonl(self) -> str:
|
|
151
|
+
return "\n".join(
|
|
152
|
+
json.dumps(record, sort_keys=True, default=str)
|
|
153
|
+
for record in self.to_records()
|
|
154
|
+
)
|
|
155
|
+
|
|
156
|
+
def write_jsonl(self, path: str | Path) -> Path:
|
|
157
|
+
target = Path(path)
|
|
158
|
+
text = self.to_jsonl()
|
|
159
|
+
target.write_text(text + ("\n" if text else ""))
|
|
160
|
+
return target
|
|
161
|
+
|
|
162
|
+
def to_futureagi_rows(self) -> list[dict[str, Any]]:
|
|
163
|
+
return [case.to_futureagi_row() for case in self.cases]
|
|
164
|
+
|
|
165
|
+
def coverage_report(
|
|
166
|
+
self,
|
|
167
|
+
*,
|
|
168
|
+
target: Any = None,
|
|
169
|
+
metric_path_hints: Optional[Mapping[str, Sequence[str]]] = None,
|
|
170
|
+
tag_path_hints: Optional[Mapping[str, Sequence[str]]] = None,
|
|
171
|
+
metadata: Optional[Mapping[str, Any]] = None,
|
|
172
|
+
) -> "AgentRegressionDatasetCoverageReport":
|
|
173
|
+
return build_agent_regression_dataset_coverage_report(
|
|
174
|
+
self,
|
|
175
|
+
target=target,
|
|
176
|
+
metric_path_hints=metric_path_hints,
|
|
177
|
+
tag_path_hints=tag_path_hints,
|
|
178
|
+
metadata=metadata,
|
|
179
|
+
)
|
|
180
|
+
|
|
181
|
+
def to_observability_window(
|
|
182
|
+
self,
|
|
183
|
+
*,
|
|
184
|
+
candidate: Optional[AgentCandidate] = None,
|
|
185
|
+
source: Optional[str] = None,
|
|
186
|
+
framework: Optional[str] = None,
|
|
187
|
+
required_metrics: Optional[Mapping[str, float]] = None,
|
|
188
|
+
required_trace_signals: Optional[Sequence[str]] = None,
|
|
189
|
+
metadata: Optional[Mapping[str, Any]] = None,
|
|
190
|
+
) -> AgentObservabilityWindow:
|
|
191
|
+
"""
|
|
192
|
+
Reconstruct an observability window from replay cases.
|
|
193
|
+
|
|
194
|
+
This is the bridge for Future AGI dataset-driven re-optimization: rows
|
|
195
|
+
pulled from a Future AGI regression dataset become the same live
|
|
196
|
+
evaluation evidence consumed by AgentFeedbackOptimizer.
|
|
197
|
+
"""
|
|
198
|
+
|
|
199
|
+
thresholds = _regression_dataset_required_metrics(
|
|
200
|
+
self.cases,
|
|
201
|
+
override=required_metrics,
|
|
202
|
+
)
|
|
203
|
+
signals = _regression_dataset_required_trace_signals(
|
|
204
|
+
self.cases,
|
|
205
|
+
override=required_trace_signals,
|
|
206
|
+
)
|
|
207
|
+
records = [
|
|
208
|
+
_observability_record_from_regression_case(
|
|
209
|
+
case,
|
|
210
|
+
index=index,
|
|
211
|
+
candidate=candidate,
|
|
212
|
+
source=source or self.source,
|
|
213
|
+
framework=framework or self.framework,
|
|
214
|
+
)
|
|
215
|
+
for index, case in enumerate(self.cases, start=1)
|
|
216
|
+
]
|
|
217
|
+
window_metadata = {
|
|
218
|
+
"kind": "regression_dataset_replay",
|
|
219
|
+
"regression_dataset_name": self.name,
|
|
220
|
+
"regression_dataset_schema_version": self.schema_version,
|
|
221
|
+
"regression_case_count": len(self.cases),
|
|
222
|
+
"regression_dataset_metadata": copy.deepcopy(self.metadata),
|
|
223
|
+
**dict(metadata or {}),
|
|
224
|
+
}
|
|
225
|
+
return AgentObservabilityWindow(
|
|
226
|
+
source=_resolve_window_source(records, fallback=source or self.source),
|
|
227
|
+
framework=_resolve_window_framework(records, fallback=framework or self.framework),
|
|
228
|
+
candidate=candidate,
|
|
229
|
+
records=records,
|
|
230
|
+
required_metrics=thresholds,
|
|
231
|
+
required_trace_signals=signals,
|
|
232
|
+
metadata=window_metadata,
|
|
233
|
+
)
|
|
234
|
+
|
|
235
|
+
def to_manifest(self) -> dict[str, Any]:
|
|
236
|
+
return self.model_dump()
|
|
237
|
+
|
|
238
|
+
def to_json(self, *, indent: int = 2) -> str:
|
|
239
|
+
return json.dumps(self.to_manifest(), sort_keys=True, indent=indent, default=str)
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
class AgentRegressionDatasetCoverageReport(BaseModel):
|
|
243
|
+
"""Coverage summary for a regression dataset before optimizer replay."""
|
|
244
|
+
|
|
245
|
+
schema_version: str = REGRESSION_DATASET_COVERAGE_SCHEMA_VERSION
|
|
246
|
+
dataset_name: str
|
|
247
|
+
source: str
|
|
248
|
+
framework: str
|
|
249
|
+
case_count: int
|
|
250
|
+
failed_case_count: int = 0
|
|
251
|
+
passed_case_count: int = 0
|
|
252
|
+
source_counts: dict[str, int] = Field(default_factory=dict)
|
|
253
|
+
framework_counts: dict[str, int] = Field(default_factory=dict)
|
|
254
|
+
tag_counts: dict[str, int] = Field(default_factory=dict)
|
|
255
|
+
observed_metric_case_counts: dict[str, int] = Field(default_factory=dict)
|
|
256
|
+
required_metric_case_counts: dict[str, int] = Field(default_factory=dict)
|
|
257
|
+
failed_metric_case_counts: dict[str, int] = Field(default_factory=dict)
|
|
258
|
+
required_metrics: dict[str, float] = Field(default_factory=dict)
|
|
259
|
+
required_trace_signals: list[str] = Field(default_factory=list)
|
|
260
|
+
trace_signal_case_counts: dict[str, int] = Field(default_factory=dict)
|
|
261
|
+
missing_trace_signal_case_counts: dict[str, int] = Field(default_factory=dict)
|
|
262
|
+
search_path_case_counts: dict[str, int] = Field(default_factory=dict)
|
|
263
|
+
uncovered_required_metrics: list[str] = Field(default_factory=list)
|
|
264
|
+
uncovered_search_paths: list[str] = Field(default_factory=list)
|
|
265
|
+
failure_examples: dict[str, list[str]] = Field(default_factory=dict)
|
|
266
|
+
metadata: dict[str, Any] = Field(default_factory=dict)
|
|
267
|
+
|
|
268
|
+
def to_manifest(self) -> dict[str, Any]:
|
|
269
|
+
return self.model_dump()
|
|
270
|
+
|
|
271
|
+
def to_json(self, *, indent: int = 2) -> str:
|
|
272
|
+
return json.dumps(self.to_manifest(), sort_keys=True, indent=indent, default=str)
|
|
273
|
+
|
|
274
|
+
|
|
275
|
+
class AgentDatasetSinkResult(BaseModel):
|
|
276
|
+
"""Result from exporting a regression dataset to Future AGI."""
|
|
277
|
+
|
|
278
|
+
schema_version: str = DATASET_SINK_SCHEMA_VERSION
|
|
279
|
+
provider: str
|
|
280
|
+
dataset_name: str
|
|
281
|
+
dataset_id: Optional[str] = None
|
|
282
|
+
case_count: int
|
|
283
|
+
endpoint: Optional[str] = None
|
|
284
|
+
dry_run: bool = False
|
|
285
|
+
status: str
|
|
286
|
+
failures: list[str] = Field(default_factory=list)
|
|
287
|
+
response: dict[str, Any] = Field(default_factory=dict)
|
|
288
|
+
metadata: dict[str, Any] = Field(default_factory=dict)
|
|
289
|
+
|
|
290
|
+
@property
|
|
291
|
+
def ok(self) -> bool:
|
|
292
|
+
return not self.failures and self.status in {"planned", "created", "inserted"}
|
|
293
|
+
|
|
294
|
+
|
|
295
|
+
class AgentRegistryReplayPackManifest(BaseModel):
|
|
296
|
+
"""Version-pinned manifest for a Future AGI registry replay pack."""
|
|
297
|
+
|
|
298
|
+
schema_version: str = REGISTRY_REPLAY_PACK_MANIFEST_SCHEMA_VERSION
|
|
299
|
+
name: str
|
|
300
|
+
provider: str = "futureagi"
|
|
301
|
+
registry_version: str
|
|
302
|
+
dataset_name: str
|
|
303
|
+
dataset_id: Optional[str] = None
|
|
304
|
+
case_count: int
|
|
305
|
+
case_ids: list[str] = Field(default_factory=list)
|
|
306
|
+
case_signature: str
|
|
307
|
+
retention_key: str
|
|
308
|
+
selection_complete: bool = False
|
|
309
|
+
coverage_score: float = 0.0
|
|
310
|
+
selected_positive_count: int = 0
|
|
311
|
+
selected_negative_count: int = 0
|
|
312
|
+
required_presets: list[str] = Field(default_factory=list)
|
|
313
|
+
required_invariant_families: list[str] = Field(default_factory=list)
|
|
314
|
+
metadata: dict[str, Any] = Field(default_factory=dict)
|
|
315
|
+
|
|
316
|
+
def to_manifest(self) -> dict[str, Any]:
|
|
317
|
+
return self.model_dump()
|
|
318
|
+
|
|
319
|
+
def to_json(self, *, indent: int = 2) -> str:
|
|
320
|
+
return json.dumps(self.to_manifest(), sort_keys=True, indent=indent, default=str)
|
|
321
|
+
|
|
322
|
+
|
|
323
|
+
class AgentRegistryReplayPackPromotionCheck(BaseModel):
|
|
324
|
+
"""Promotion gate for a pinned Future AGI registry replay pack."""
|
|
325
|
+
|
|
326
|
+
schema_version: str = REGISTRY_REPLAY_PACK_PROMOTION_SCHEMA_VERSION
|
|
327
|
+
promotable: bool
|
|
328
|
+
dataset_name: str
|
|
329
|
+
dataset_id: Optional[str] = None
|
|
330
|
+
registry_version: str
|
|
331
|
+
expected_registry_version: str
|
|
332
|
+
expected_case_count: int
|
|
333
|
+
loaded_case_count: int
|
|
334
|
+
expected_case_signature: str
|
|
335
|
+
loaded_case_signature: str
|
|
336
|
+
coverage_score: float
|
|
337
|
+
min_coverage_score: float
|
|
338
|
+
selection_complete: bool
|
|
339
|
+
replay_record_count: int = 0
|
|
340
|
+
optimizer_score: Optional[float] = None
|
|
341
|
+
min_optimizer_score: float
|
|
342
|
+
failures: list[str] = Field(default_factory=list)
|
|
343
|
+
manifest: AgentRegistryReplayPackManifest
|
|
344
|
+
metadata: dict[str, Any] = Field(default_factory=dict)
|
|
345
|
+
|
|
346
|
+
def to_manifest(self) -> dict[str, Any]:
|
|
347
|
+
return self.model_dump()
|
|
348
|
+
|
|
349
|
+
def to_json(self, *, indent: int = 2) -> str:
|
|
350
|
+
return json.dumps(self.to_manifest(), sort_keys=True, indent=indent, default=str)
|
|
351
|
+
|
|
352
|
+
|
|
353
|
+
class AgentRegistryReplayPackLineageEntry(BaseModel):
|
|
354
|
+
"""One versioned Future AGI registry replay-pack lineage row."""
|
|
355
|
+
|
|
356
|
+
registry_version: str
|
|
357
|
+
dataset_name: str
|
|
358
|
+
dataset_id: Optional[str] = None
|
|
359
|
+
retention_key: str
|
|
360
|
+
case_count: int
|
|
361
|
+
case_signature: str
|
|
362
|
+
coverage_score: float
|
|
363
|
+
selection_complete: bool
|
|
364
|
+
required_presets: list[str] = Field(default_factory=list)
|
|
365
|
+
required_invariant_families: list[str] = Field(default_factory=list)
|
|
366
|
+
promotion_promotable: Optional[bool] = None
|
|
367
|
+
loaded_case_count: Optional[int] = None
|
|
368
|
+
replay_record_count: Optional[int] = None
|
|
369
|
+
readback_signature_matches: Optional[bool] = None
|
|
370
|
+
optimizer_score: Optional[float] = None
|
|
371
|
+
min_optimizer_score: Optional[float] = None
|
|
372
|
+
optimizer_backend: Optional[str] = None
|
|
373
|
+
selected_patch_signature: Optional[str] = None
|
|
374
|
+
failures: list[str] = Field(default_factory=list)
|
|
375
|
+
metadata: dict[str, Any] = Field(default_factory=dict)
|
|
376
|
+
|
|
377
|
+
|
|
378
|
+
class AgentRegistryReplayPackLineageTransition(BaseModel):
|
|
379
|
+
"""Delta from one registry replay-pack version to the next."""
|
|
380
|
+
|
|
381
|
+
from_registry_version: str
|
|
382
|
+
to_registry_version: str
|
|
383
|
+
from_dataset_id: Optional[str] = None
|
|
384
|
+
to_dataset_id: Optional[str] = None
|
|
385
|
+
case_count_delta: int = 0
|
|
386
|
+
coverage_delta: float = 0.0
|
|
387
|
+
optimizer_score_delta: Optional[float] = None
|
|
388
|
+
case_signature_changed: bool = False
|
|
389
|
+
retention_key_changed: bool = False
|
|
390
|
+
selected_patch_changed: Optional[bool] = None
|
|
391
|
+
optimizer_backend_changed: Optional[bool] = None
|
|
392
|
+
promotion_status_changed: Optional[bool] = None
|
|
393
|
+
added_required_presets: list[str] = Field(default_factory=list)
|
|
394
|
+
removed_required_presets: list[str] = Field(default_factory=list)
|
|
395
|
+
added_invariant_families: list[str] = Field(default_factory=list)
|
|
396
|
+
removed_invariant_families: list[str] = Field(default_factory=list)
|
|
397
|
+
drift_reasons: list[str] = Field(default_factory=list)
|
|
398
|
+
|
|
399
|
+
|
|
400
|
+
class AgentRegistryReplayPackLineageReport(BaseModel):
|
|
401
|
+
"""Compare Future AGI registry replay packs across dataset versions."""
|
|
402
|
+
|
|
403
|
+
schema_version: str = REGISTRY_REPLAY_PACK_LINEAGE_SCHEMA_VERSION
|
|
404
|
+
provider: str = "futureagi"
|
|
405
|
+
entry_count: int
|
|
406
|
+
entries: list[AgentRegistryReplayPackLineageEntry] = Field(default_factory=list)
|
|
407
|
+
transitions: list[AgentRegistryReplayPackLineageTransition] = Field(default_factory=list)
|
|
408
|
+
latest_registry_version: Optional[str] = None
|
|
409
|
+
latest_dataset_id: Optional[str] = None
|
|
410
|
+
latest_promotable: Optional[bool] = None
|
|
411
|
+
best_registry_version: Optional[str] = None
|
|
412
|
+
best_dataset_id: Optional[str] = None
|
|
413
|
+
best_optimizer_score: Optional[float] = None
|
|
414
|
+
drift_detected: bool = False
|
|
415
|
+
drift_reasons: list[str] = Field(default_factory=list)
|
|
416
|
+
metadata: dict[str, Any] = Field(default_factory=dict)
|
|
417
|
+
|
|
418
|
+
def to_manifest(self) -> dict[str, Any]:
|
|
419
|
+
return self.model_dump()
|
|
420
|
+
|
|
421
|
+
def to_json(self, *, indent: int = 2) -> str:
|
|
422
|
+
return json.dumps(self.to_manifest(), sort_keys=True, indent=indent, default=str)
|
|
423
|
+
|
|
424
|
+
|
|
425
|
+
class AgentRegistryReplayPackTriageReport(BaseModel):
|
|
426
|
+
"""Rollout triage for Future AGI registry replay-pack lineage drift."""
|
|
427
|
+
|
|
428
|
+
schema_version: str = REGISTRY_REPLAY_PACK_TRIAGE_SCHEMA_VERSION
|
|
429
|
+
provider: str = "futureagi"
|
|
430
|
+
decision: str
|
|
431
|
+
severity: str
|
|
432
|
+
block_rollout: bool
|
|
433
|
+
latest_registry_version: Optional[str] = None
|
|
434
|
+
latest_dataset_id: Optional[str] = None
|
|
435
|
+
baseline_registry_version: Optional[str] = None
|
|
436
|
+
baseline_dataset_id: Optional[str] = None
|
|
437
|
+
best_registry_version: Optional[str] = None
|
|
438
|
+
best_dataset_id: Optional[str] = None
|
|
439
|
+
latest_promotable: Optional[bool] = None
|
|
440
|
+
coverage_delta: Optional[float] = None
|
|
441
|
+
optimizer_score_delta: Optional[float] = None
|
|
442
|
+
best_optimizer_score_gap: Optional[float] = None
|
|
443
|
+
blocking_reasons: list[str] = Field(default_factory=list)
|
|
444
|
+
warnings: list[str] = Field(default_factory=list)
|
|
445
|
+
drift_reasons: list[str] = Field(default_factory=list)
|
|
446
|
+
recommendations: list[str] = Field(default_factory=list)
|
|
447
|
+
evidence: dict[str, Any] = Field(default_factory=dict)
|
|
448
|
+
metadata: dict[str, Any] = Field(default_factory=dict)
|
|
449
|
+
|
|
450
|
+
def to_manifest(self) -> dict[str, Any]:
|
|
451
|
+
return self.model_dump()
|
|
452
|
+
|
|
453
|
+
def to_json(self, *, indent: int = 2) -> str:
|
|
454
|
+
return json.dumps(self.to_manifest(), sort_keys=True, indent=indent, default=str)
|
|
455
|
+
|
|
456
|
+
|
|
457
|
+
def load_agent_observability_feedback(
|
|
458
|
+
payload: Any,
|
|
459
|
+
*,
|
|
460
|
+
candidate: Optional[AgentCandidate] = None,
|
|
461
|
+
source: str = "auto",
|
|
462
|
+
framework: str = "auto",
|
|
463
|
+
required_metrics: Optional[Mapping[str, float]] = None,
|
|
464
|
+
required_trace_signals: Optional[Sequence[str]] = None,
|
|
465
|
+
metadata: Optional[Mapping[str, Any]] = None,
|
|
466
|
+
) -> AgentObservabilityWindow:
|
|
467
|
+
"""
|
|
468
|
+
Normalize production observability exports into agent optimization feedback.
|
|
469
|
+
|
|
470
|
+
Accepted payloads include generic run/feedback exports, OpenAI Agents trace
|
|
471
|
+
processor exports, OpenTelemetry/TraceAI OTLP JSON, LiveKit session reports,
|
|
472
|
+
plain JSON/JSONL strings, file paths, or lists of those records.
|
|
473
|
+
"""
|
|
474
|
+
|
|
475
|
+
loaded = _load_payload(payload)
|
|
476
|
+
raw_records = _observation_records(loaded)
|
|
477
|
+
records: list[AgentObservabilityRecord] = []
|
|
478
|
+
thresholds = {
|
|
479
|
+
str(key): float(value)
|
|
480
|
+
for key, value in dict(required_metrics or {}).items()
|
|
481
|
+
}
|
|
482
|
+
required_signals = [_normalize_signal(item) for item in required_trace_signals or []]
|
|
483
|
+
required_signals = [item for item in required_signals if item]
|
|
484
|
+
for index, raw_record in enumerate(raw_records, start=1):
|
|
485
|
+
if not isinstance(raw_record, Mapping):
|
|
486
|
+
raw_record = {"value": raw_record}
|
|
487
|
+
records.append(
|
|
488
|
+
_normalize_observability_record(
|
|
489
|
+
dict(raw_record),
|
|
490
|
+
index=index,
|
|
491
|
+
candidate=candidate,
|
|
492
|
+
source=source,
|
|
493
|
+
framework=framework,
|
|
494
|
+
required_metrics=thresholds,
|
|
495
|
+
required_trace_signals=required_signals,
|
|
496
|
+
)
|
|
497
|
+
)
|
|
498
|
+
|
|
499
|
+
return AgentObservabilityWindow(
|
|
500
|
+
source=_resolve_window_source(records, fallback=source),
|
|
501
|
+
framework=_resolve_window_framework(records, fallback=framework),
|
|
502
|
+
candidate=candidate,
|
|
503
|
+
records=records,
|
|
504
|
+
required_metrics=thresholds,
|
|
505
|
+
required_trace_signals=required_signals,
|
|
506
|
+
metadata=dict(metadata or {}),
|
|
507
|
+
)
|
|
508
|
+
|
|
509
|
+
|
|
510
|
+
def load_agent_report_replay_cases(
|
|
511
|
+
payload: Any,
|
|
512
|
+
*,
|
|
513
|
+
candidate: Optional[AgentCandidate] = None,
|
|
514
|
+
source: str = "futureagi",
|
|
515
|
+
framework: str = "agent_report",
|
|
516
|
+
required_metrics: Optional[Mapping[str, float]] = None,
|
|
517
|
+
metadata: Optional[Mapping[str, Any]] = None,
|
|
518
|
+
) -> AgentObservabilityWindow:
|
|
519
|
+
"""
|
|
520
|
+
Normalize replay cases with attached agent-report evaluations.
|
|
521
|
+
|
|
522
|
+
This bridges deterministic replay packs, including domain-package registry
|
|
523
|
+
mutation cases, into the same observability window consumed by regression
|
|
524
|
+
dataset export and feedback optimizers. No hosted service is called here.
|
|
525
|
+
"""
|
|
526
|
+
|
|
527
|
+
loaded = _load_payload(payload)
|
|
528
|
+
raw_cases = _observation_records(loaded)
|
|
529
|
+
thresholds = {str(key): float(value) for key, value in dict(required_metrics or {}).items()}
|
|
530
|
+
records: list[AgentObservabilityRecord] = []
|
|
531
|
+
for index, raw_case in enumerate(raw_cases, start=1):
|
|
532
|
+
case = _ensure_mapping(raw_case)
|
|
533
|
+
expected = _ensure_mapping(case.get("expected") or case.get("expected_response"))
|
|
534
|
+
case_thresholds = _float_mapping(expected.get("required_metrics"))
|
|
535
|
+
for name, threshold in case_thresholds.items():
|
|
536
|
+
thresholds.setdefault(name, threshold)
|
|
537
|
+
raw_evidence = _agent_report_case_raw_evidence(case)
|
|
538
|
+
evaluation = (
|
|
539
|
+
raw_evidence.get("agent_report_evaluation")
|
|
540
|
+
or raw_evidence.get("evaluation")
|
|
541
|
+
or case.get("agent_report_evaluation")
|
|
542
|
+
)
|
|
543
|
+
metrics = _agent_report_evaluation_metrics(evaluation)
|
|
544
|
+
if not metrics:
|
|
545
|
+
metrics = _float_mapping(raw_evidence.get("metrics") or case.get("metrics"))
|
|
546
|
+
active_thresholds = dict(thresholds or case_thresholds)
|
|
547
|
+
failures = _agent_report_evaluation_failures(evaluation)
|
|
548
|
+
failures.extend(_string_list(raw_evidence.get("failures") or case.get("failures")))
|
|
549
|
+
failures.extend(_metric_threshold_failures(metrics, active_thresholds))
|
|
550
|
+
failures = list(dict.fromkeys(failures))
|
|
551
|
+
evaluation_payload = _ensure_mapping(evaluation)
|
|
552
|
+
score = _coerce_score(evaluation_payload.get("score"))
|
|
553
|
+
if score is None:
|
|
554
|
+
score = min(metrics.values()) if metrics else (0.0 if failures else 1.0)
|
|
555
|
+
passed_value = evaluation_payload.get("passed")
|
|
556
|
+
if isinstance(passed_value, bool):
|
|
557
|
+
passed = passed_value and not _metric_threshold_failures(metrics, active_thresholds)
|
|
558
|
+
else:
|
|
559
|
+
passed = not failures
|
|
560
|
+
case_id = str(case.get("id") or case.get("case_id") or f"agent_report_case_{index}")
|
|
561
|
+
raw_payload = {
|
|
562
|
+
"case": copy.deepcopy(case),
|
|
563
|
+
**copy.deepcopy(raw_evidence),
|
|
564
|
+
}
|
|
565
|
+
if evaluation and "agent_report_evaluation" not in raw_payload:
|
|
566
|
+
raw_payload["agent_report_evaluation"] = copy.deepcopy(evaluation)
|
|
567
|
+
records.append(
|
|
568
|
+
AgentObservabilityRecord(
|
|
569
|
+
index=index,
|
|
570
|
+
source=_normalize_source(source),
|
|
571
|
+
framework=_normalize_source(framework),
|
|
572
|
+
run_id=case_id,
|
|
573
|
+
candidate_id=candidate.id if candidate is not None else None,
|
|
574
|
+
score=float(score),
|
|
575
|
+
passed=bool(passed),
|
|
576
|
+
failures=failures,
|
|
577
|
+
metrics=metrics,
|
|
578
|
+
raw=raw_payload,
|
|
579
|
+
metadata={
|
|
580
|
+
"source_kind": _normalize_source(source),
|
|
581
|
+
"framework": _normalize_source(framework),
|
|
582
|
+
"case_id": case_id,
|
|
583
|
+
"case_metadata": copy.deepcopy(_ensure_mapping(case.get("metadata"))),
|
|
584
|
+
},
|
|
585
|
+
)
|
|
586
|
+
)
|
|
587
|
+
return AgentObservabilityWindow(
|
|
588
|
+
source=_normalize_source(source),
|
|
589
|
+
framework=_normalize_source(framework),
|
|
590
|
+
candidate=candidate,
|
|
591
|
+
records=records,
|
|
592
|
+
required_metrics=thresholds,
|
|
593
|
+
metadata={
|
|
594
|
+
"kind": "agent_report_replay_cases",
|
|
595
|
+
"case_count": len(records),
|
|
596
|
+
**dict(metadata or {}),
|
|
597
|
+
},
|
|
598
|
+
)
|
|
599
|
+
|
|
600
|
+
|
|
601
|
+
def publish_futureagi_regression_dataset(
|
|
602
|
+
dataset: AgentRegressionDataset,
|
|
603
|
+
*,
|
|
604
|
+
dataset_name: Optional[str] = None,
|
|
605
|
+
dataset_id: Optional[str] = None,
|
|
606
|
+
description: Optional[str] = None,
|
|
607
|
+
fi_api_key: Optional[str] = None,
|
|
608
|
+
fi_secret_key: Optional[str] = None,
|
|
609
|
+
fi_base_url: Optional[str] = None,
|
|
610
|
+
dry_run: bool = False,
|
|
611
|
+
client: Any = None,
|
|
612
|
+
timeout: float = 30.0,
|
|
613
|
+
metadata: Optional[Mapping[str, Any]] = None,
|
|
614
|
+
) -> AgentDatasetSinkResult:
|
|
615
|
+
"""
|
|
616
|
+
Publish regression cases to a Future AGI dataset.
|
|
617
|
+
|
|
618
|
+
The preferred path uses the Future AGI `fi.datasets.Dataset` SDK. If that
|
|
619
|
+
SDK surface is unavailable, the publisher falls back to the Future AGI HTTP
|
|
620
|
+
dataset API through `ai-evaluation` auth primitives. Tests may inject a
|
|
621
|
+
Future AGI-compatible client with `publish_regression_dataset()`,
|
|
622
|
+
`create_regression_dataset()`, or `create_dataset()`.
|
|
623
|
+
"""
|
|
624
|
+
active_name = dataset_name or dataset.name
|
|
625
|
+
columns = _futureagi_dataset_columns()
|
|
626
|
+
rows = dataset.to_futureagi_rows()
|
|
627
|
+
active_endpoint = (
|
|
628
|
+
fi_base_url or os.getenv("FI_BASE_URL") or "https://api.futureagi.com"
|
|
629
|
+
).rstrip("/")
|
|
630
|
+
result_metadata = {
|
|
631
|
+
"description": description,
|
|
632
|
+
**dict(metadata or {}),
|
|
633
|
+
}
|
|
634
|
+
if dry_run:
|
|
635
|
+
return _dataset_sink_result(
|
|
636
|
+
provider="futureagi",
|
|
637
|
+
dataset_name=active_name,
|
|
638
|
+
dataset_id=dataset_id,
|
|
639
|
+
case_count=len(rows),
|
|
640
|
+
endpoint=active_endpoint,
|
|
641
|
+
dry_run=True,
|
|
642
|
+
status="planned",
|
|
643
|
+
response={"columns": columns, "rows": rows},
|
|
644
|
+
metadata=result_metadata,
|
|
645
|
+
)
|
|
646
|
+
|
|
647
|
+
if client is None:
|
|
648
|
+
active_api_key = fi_api_key or os.getenv("FI_API_KEY")
|
|
649
|
+
active_secret_key = fi_secret_key or os.getenv("FI_SECRET_KEY")
|
|
650
|
+
if not active_api_key or not active_secret_key:
|
|
651
|
+
return _dataset_sink_result(
|
|
652
|
+
provider="futureagi",
|
|
653
|
+
dataset_name=active_name,
|
|
654
|
+
dataset_id=dataset_id,
|
|
655
|
+
case_count=len(rows),
|
|
656
|
+
endpoint=active_endpoint,
|
|
657
|
+
status="failed",
|
|
658
|
+
failures=[
|
|
659
|
+
"Future AGI publishing requires FI_API_KEY and FI_SECRET_KEY "
|
|
660
|
+
"or an injected Future AGI dataset client."
|
|
661
|
+
],
|
|
662
|
+
metadata=result_metadata,
|
|
663
|
+
)
|
|
664
|
+
client = _load_futureagi_dataset_client(
|
|
665
|
+
fi_api_key=active_api_key,
|
|
666
|
+
fi_secret_key=active_secret_key,
|
|
667
|
+
fi_base_url=active_endpoint,
|
|
668
|
+
timeout=timeout,
|
|
669
|
+
)
|
|
670
|
+
if client is None:
|
|
671
|
+
return _dataset_sink_result(
|
|
672
|
+
provider="futureagi",
|
|
673
|
+
dataset_name=active_name,
|
|
674
|
+
dataset_id=dataset_id,
|
|
675
|
+
case_count=len(rows),
|
|
676
|
+
endpoint=active_endpoint,
|
|
677
|
+
status="failed",
|
|
678
|
+
failures=[
|
|
679
|
+
"Future AGI publishing requires the `futureagi` dataset SDK "
|
|
680
|
+
"or the `ai-evaluation` HTTP auth primitives."
|
|
681
|
+
],
|
|
682
|
+
metadata=result_metadata,
|
|
683
|
+
)
|
|
684
|
+
|
|
685
|
+
try:
|
|
686
|
+
response = _publish_futureagi_regression_dataset_with_client(
|
|
687
|
+
client,
|
|
688
|
+
dataset_name=active_name,
|
|
689
|
+
dataset_id=dataset_id,
|
|
690
|
+
columns=columns,
|
|
691
|
+
rows=rows,
|
|
692
|
+
metadata=result_metadata,
|
|
693
|
+
)
|
|
694
|
+
except Exception as exc:
|
|
695
|
+
return _dataset_sink_result(
|
|
696
|
+
provider="futureagi",
|
|
697
|
+
dataset_name=active_name,
|
|
698
|
+
dataset_id=dataset_id,
|
|
699
|
+
case_count=len(rows),
|
|
700
|
+
endpoint=active_endpoint,
|
|
701
|
+
status="failed",
|
|
702
|
+
failures=[f"Future AGI dataset publish failed: {exc}"],
|
|
703
|
+
metadata=result_metadata,
|
|
704
|
+
)
|
|
705
|
+
|
|
706
|
+
response_payload = _safe_response_payload(response)
|
|
707
|
+
return _dataset_sink_result(
|
|
708
|
+
provider="futureagi",
|
|
709
|
+
dataset_name=active_name,
|
|
710
|
+
dataset_id=_response_id(response) or _response_id(response_payload) or dataset_id,
|
|
711
|
+
case_count=len(rows),
|
|
712
|
+
endpoint=active_endpoint,
|
|
713
|
+
status="created" if dataset_id is None else "inserted",
|
|
714
|
+
response=response_payload,
|
|
715
|
+
metadata=result_metadata,
|
|
716
|
+
)
|
|
717
|
+
|
|
718
|
+
|
|
719
|
+
def load_futureagi_regression_dataset(
|
|
720
|
+
*,
|
|
721
|
+
dataset_id: str,
|
|
722
|
+
dataset_name: Optional[str] = None,
|
|
723
|
+
fi_api_key: Optional[str] = None,
|
|
724
|
+
fi_secret_key: Optional[str] = None,
|
|
725
|
+
fi_base_url: Optional[str] = None,
|
|
726
|
+
client: Any = None,
|
|
727
|
+
page_size: int = 100,
|
|
728
|
+
max_pages: int = 100,
|
|
729
|
+
timeout: float = 30.0,
|
|
730
|
+
metadata: Optional[Mapping[str, Any]] = None,
|
|
731
|
+
) -> AgentRegressionDataset:
|
|
732
|
+
"""
|
|
733
|
+
Pull a Future AGI regression dataset back into replayable agent-opt cases.
|
|
734
|
+
|
|
735
|
+
The loader reads the Future AGI dataset table API, maps column ids back to
|
|
736
|
+
`publish_futureagi_regression_dataset()` column names, parses JSON/array
|
|
737
|
+
cells, and reconstructs an `AgentRegressionDataset` that can be converted
|
|
738
|
+
to an `AgentObservabilityWindow` for metric-based re-optimization.
|
|
739
|
+
"""
|
|
740
|
+
|
|
741
|
+
if not dataset_id:
|
|
742
|
+
raise ValueError("load_futureagi_regression_dataset requires dataset_id.")
|
|
743
|
+
if page_size < 1:
|
|
744
|
+
raise ValueError("page_size must be at least 1.")
|
|
745
|
+
if max_pages < 1:
|
|
746
|
+
raise ValueError("max_pages must be at least 1.")
|
|
747
|
+
|
|
748
|
+
active_endpoint = (fi_base_url or os.getenv("FI_BASE_URL") or "https://api.futureagi.com").rstrip("/")
|
|
749
|
+
if client is None:
|
|
750
|
+
active_api_key = fi_api_key or os.getenv("FI_API_KEY")
|
|
751
|
+
active_secret_key = fi_secret_key or os.getenv("FI_SECRET_KEY")
|
|
752
|
+
if not active_api_key or not active_secret_key:
|
|
753
|
+
raise ValueError(
|
|
754
|
+
"Future AGI dataset loading requires FI_API_KEY and FI_SECRET_KEY "
|
|
755
|
+
"or an injected Future AGI dataset client."
|
|
756
|
+
)
|
|
757
|
+
client = _load_futureagi_dataset_reader_client(
|
|
758
|
+
fi_api_key=active_api_key,
|
|
759
|
+
fi_secret_key=active_secret_key,
|
|
760
|
+
fi_base_url=active_endpoint,
|
|
761
|
+
timeout=timeout,
|
|
762
|
+
)
|
|
763
|
+
if client is None:
|
|
764
|
+
raise RuntimeError(
|
|
765
|
+
"Future AGI dataset loading requires the `ai-evaluation` HTTP "
|
|
766
|
+
"auth primitives."
|
|
767
|
+
)
|
|
768
|
+
|
|
769
|
+
payloads = _futureagi_dataset_payloads(
|
|
770
|
+
client,
|
|
771
|
+
dataset_id=dataset_id,
|
|
772
|
+
page_size=page_size,
|
|
773
|
+
max_pages=max_pages,
|
|
774
|
+
)
|
|
775
|
+
cases, table_metadata = _futureagi_regression_cases_from_payloads(
|
|
776
|
+
payloads,
|
|
777
|
+
dataset_id=dataset_id,
|
|
778
|
+
)
|
|
779
|
+
resolved_name = (
|
|
780
|
+
dataset_name
|
|
781
|
+
or str(table_metadata.get("dataset_name") or "").strip()
|
|
782
|
+
or f"futureagi-regression-{dataset_id}"
|
|
783
|
+
)
|
|
784
|
+
return AgentRegressionDataset(
|
|
785
|
+
name=resolved_name,
|
|
786
|
+
source=_regression_cases_source(cases),
|
|
787
|
+
framework=_regression_cases_framework(cases),
|
|
788
|
+
cases=cases,
|
|
789
|
+
metadata={
|
|
790
|
+
"kind": "futureagi_regression_dataset",
|
|
791
|
+
"dataset_id": dataset_id,
|
|
792
|
+
"endpoint": active_endpoint,
|
|
793
|
+
"page_count": len(payloads),
|
|
794
|
+
"row_count": len(cases),
|
|
795
|
+
"column_count": int(table_metadata.get("column_count") or 0),
|
|
796
|
+
"futureagi_metadata": {
|
|
797
|
+
key: value
|
|
798
|
+
for key, value in table_metadata.items()
|
|
799
|
+
if key != "column_count"
|
|
800
|
+
},
|
|
801
|
+
**dict(metadata or {}),
|
|
802
|
+
},
|
|
803
|
+
)
|
|
804
|
+
|
|
805
|
+
|
|
806
|
+
def build_futureagi_registry_replay_pack_manifest(
|
|
807
|
+
dataset: AgentRegressionDataset,
|
|
808
|
+
*,
|
|
809
|
+
publish_result: Optional[AgentDatasetSinkResult] = None,
|
|
810
|
+
registry_version: Optional[str] = None,
|
|
811
|
+
selection: Optional[Mapping[str, Any]] = None,
|
|
812
|
+
name: Optional[str] = None,
|
|
813
|
+
metadata: Optional[Mapping[str, Any]] = None,
|
|
814
|
+
) -> AgentRegistryReplayPackManifest:
|
|
815
|
+
"""
|
|
816
|
+
Build a version-pinned manifest for a Future AGI registry replay pack.
|
|
817
|
+
|
|
818
|
+
The manifest is a retention record: it pins the registry version, Future AGI
|
|
819
|
+
dataset id/name, selected case ids, coverage signal, and a deterministic
|
|
820
|
+
case signature that a promotion gate can verify after live readback.
|
|
821
|
+
"""
|
|
822
|
+
|
|
823
|
+
selection_metadata = _ensure_mapping(selection)
|
|
824
|
+
result_metadata = _ensure_mapping(publish_result.metadata if publish_result else {})
|
|
825
|
+
dataset_metadata = _ensure_mapping(dataset.metadata)
|
|
826
|
+
active_registry_version = str(
|
|
827
|
+
registry_version
|
|
828
|
+
or dataset_metadata.get("registry_version")
|
|
829
|
+
or result_metadata.get("registry_version")
|
|
830
|
+
or selection_metadata.get("registry_version")
|
|
831
|
+
or ""
|
|
832
|
+
).strip()
|
|
833
|
+
if not active_registry_version:
|
|
834
|
+
raise ValueError("registry_version is required for registry replay pack manifests.")
|
|
835
|
+
|
|
836
|
+
dataset_id = (
|
|
837
|
+
publish_result.dataset_id
|
|
838
|
+
if publish_result is not None
|
|
839
|
+
else _optional_str(dataset_metadata.get("dataset_id"))
|
|
840
|
+
)
|
|
841
|
+
dataset_name = (
|
|
842
|
+
publish_result.dataset_name
|
|
843
|
+
if publish_result is not None
|
|
844
|
+
else dataset.name
|
|
845
|
+
)
|
|
846
|
+
case_ids = [str(case.id) for case in dataset.cases]
|
|
847
|
+
case_signature = _registry_replay_case_signature(case_ids)
|
|
848
|
+
required_presets, required_families = _registry_replay_requirements(selection_metadata)
|
|
849
|
+
coverage_score = _first_float(
|
|
850
|
+
selection_metadata.get("selected_coverage", {}).get("coverage_score")
|
|
851
|
+
if isinstance(selection_metadata.get("selected_coverage"), Mapping)
|
|
852
|
+
else None,
|
|
853
|
+
selection_metadata.get("coverage_score"),
|
|
854
|
+
dataset_metadata.get("coverage_score"),
|
|
855
|
+
result_metadata.get("coverage_score"),
|
|
856
|
+
0.0,
|
|
857
|
+
)
|
|
858
|
+
selection_complete = bool(
|
|
859
|
+
selection_metadata.get("selection_complete")
|
|
860
|
+
if "selection_complete" in selection_metadata
|
|
861
|
+
else dataset_metadata.get("selection_complete")
|
|
862
|
+
if "selection_complete" in dataset_metadata
|
|
863
|
+
else result_metadata.get("selection_complete", False)
|
|
864
|
+
)
|
|
865
|
+
selected_positive_count = int(
|
|
866
|
+
_first_float(
|
|
867
|
+
selection_metadata.get("selected_positive_count"),
|
|
868
|
+
dataset_metadata.get("selected_positive_count"),
|
|
869
|
+
0,
|
|
870
|
+
)
|
|
871
|
+
)
|
|
872
|
+
selected_negative_count = int(
|
|
873
|
+
_first_float(
|
|
874
|
+
selection_metadata.get("selected_negative_count"),
|
|
875
|
+
dataset_metadata.get("selected_negative_count"),
|
|
876
|
+
0,
|
|
877
|
+
)
|
|
878
|
+
)
|
|
879
|
+
provider = publish_result.provider if publish_result is not None else "futureagi"
|
|
880
|
+
manifest_name = name or f"{active_registry_version}:{dataset_name}"
|
|
881
|
+
return AgentRegistryReplayPackManifest(
|
|
882
|
+
name=manifest_name,
|
|
883
|
+
provider=str(provider or "futureagi"),
|
|
884
|
+
registry_version=active_registry_version,
|
|
885
|
+
dataset_name=str(dataset_name),
|
|
886
|
+
dataset_id=dataset_id,
|
|
887
|
+
case_count=len(dataset.cases),
|
|
888
|
+
case_ids=case_ids,
|
|
889
|
+
case_signature=case_signature,
|
|
890
|
+
retention_key=_registry_replay_retention_key(
|
|
891
|
+
registry_version=active_registry_version,
|
|
892
|
+
dataset_name=str(dataset_name),
|
|
893
|
+
dataset_id=dataset_id,
|
|
894
|
+
case_signature=case_signature,
|
|
895
|
+
),
|
|
896
|
+
selection_complete=selection_complete,
|
|
897
|
+
coverage_score=coverage_score,
|
|
898
|
+
selected_positive_count=selected_positive_count,
|
|
899
|
+
selected_negative_count=selected_negative_count,
|
|
900
|
+
required_presets=required_presets,
|
|
901
|
+
required_invariant_families=required_families,
|
|
902
|
+
metadata={
|
|
903
|
+
"dataset_source": dataset.source,
|
|
904
|
+
"dataset_framework": dataset.framework,
|
|
905
|
+
"dataset_metadata": copy.deepcopy(dataset.metadata),
|
|
906
|
+
"publish_result_status": publish_result.status if publish_result else None,
|
|
907
|
+
**dict(metadata or {}),
|
|
908
|
+
},
|
|
909
|
+
)
|
|
910
|
+
|
|
911
|
+
|
|
912
|
+
def check_futureagi_registry_replay_pack_promotion(
|
|
913
|
+
manifest: AgentRegistryReplayPackManifest | Mapping[str, Any],
|
|
914
|
+
*,
|
|
915
|
+
registry_version: Optional[str] = None,
|
|
916
|
+
dataset: Optional[AgentRegressionDataset] = None,
|
|
917
|
+
dataset_id: Optional[str] = None,
|
|
918
|
+
candidate: Optional[AgentCandidate] = None,
|
|
919
|
+
optimizer_result: Any = None,
|
|
920
|
+
optimizer_score: Optional[float] = None,
|
|
921
|
+
min_coverage_score: float = 1.0,
|
|
922
|
+
min_optimizer_score: float = 0.99,
|
|
923
|
+
fi_api_key: Optional[str] = None,
|
|
924
|
+
fi_secret_key: Optional[str] = None,
|
|
925
|
+
fi_base_url: Optional[str] = None,
|
|
926
|
+
client: Any = None,
|
|
927
|
+
page_size: int = 100,
|
|
928
|
+
max_pages: int = 100,
|
|
929
|
+
timeout: float = 30.0,
|
|
930
|
+
metadata: Optional[Mapping[str, Any]] = None,
|
|
931
|
+
) -> AgentRegistryReplayPackPromotionCheck:
|
|
932
|
+
"""
|
|
933
|
+
Gate a selected registry replay pack before promotion.
|
|
934
|
+
|
|
935
|
+
A pack is promotable only when the manifest's registry version matches the
|
|
936
|
+
expected version, the pinned Future AGI dataset reads back with the same
|
|
937
|
+
cases, coverage meets the required threshold, and optimizer replay evidence
|
|
938
|
+
reaches `min_optimizer_score`.
|
|
939
|
+
"""
|
|
940
|
+
|
|
941
|
+
if min_coverage_score < 0:
|
|
942
|
+
raise ValueError("min_coverage_score must be non-negative.")
|
|
943
|
+
if min_optimizer_score < 0:
|
|
944
|
+
raise ValueError("min_optimizer_score must be non-negative.")
|
|
945
|
+
|
|
946
|
+
active_manifest = _coerce_registry_replay_manifest(manifest)
|
|
947
|
+
expected_registry_version = str(registry_version or active_manifest.registry_version)
|
|
948
|
+
active_dataset = dataset
|
|
949
|
+
active_dataset_id = dataset_id or active_manifest.dataset_id
|
|
950
|
+
load_failure: Optional[str] = None
|
|
951
|
+
if active_dataset is None and active_dataset_id:
|
|
952
|
+
try:
|
|
953
|
+
active_dataset = load_futureagi_regression_dataset(
|
|
954
|
+
dataset_id=active_dataset_id,
|
|
955
|
+
fi_api_key=fi_api_key,
|
|
956
|
+
fi_secret_key=fi_secret_key,
|
|
957
|
+
fi_base_url=fi_base_url,
|
|
958
|
+
client=client,
|
|
959
|
+
page_size=page_size,
|
|
960
|
+
max_pages=max_pages,
|
|
961
|
+
timeout=timeout,
|
|
962
|
+
metadata={"promotion_gate": "registry_replay_pack"},
|
|
963
|
+
)
|
|
964
|
+
except Exception as exc:
|
|
965
|
+
load_failure = f"Future AGI readback failed: {exc}"
|
|
966
|
+
elif active_dataset is None:
|
|
967
|
+
load_failure = "Future AGI dataset id is required for registry replay pack promotion."
|
|
968
|
+
|
|
969
|
+
loaded_case_ids = [str(case.id) for case in active_dataset.cases] if active_dataset else []
|
|
970
|
+
loaded_case_count = len(loaded_case_ids)
|
|
971
|
+
loaded_case_signature = _registry_replay_case_signature(loaded_case_ids)
|
|
972
|
+
replay_record_count = 0
|
|
973
|
+
if active_dataset is not None:
|
|
974
|
+
replay_window = active_dataset.to_observability_window(
|
|
975
|
+
candidate=candidate,
|
|
976
|
+
source="futureagi",
|
|
977
|
+
framework=active_dataset.framework,
|
|
978
|
+
metadata={"promotion_gate": "registry_replay_pack"},
|
|
979
|
+
)
|
|
980
|
+
replay_record_count = len(replay_window.records)
|
|
981
|
+
|
|
982
|
+
active_optimizer_score = (
|
|
983
|
+
float(optimizer_score)
|
|
984
|
+
if optimizer_score is not None
|
|
985
|
+
else _optimizer_result_score(optimizer_result)
|
|
986
|
+
)
|
|
987
|
+
failures: list[str] = []
|
|
988
|
+
if load_failure:
|
|
989
|
+
failures.append(load_failure)
|
|
990
|
+
if active_manifest.provider.lower() != "futureagi":
|
|
991
|
+
failures.append(
|
|
992
|
+
f"registry replay pack provider '{active_manifest.provider}' is not Future AGI"
|
|
993
|
+
)
|
|
994
|
+
if not active_manifest.dataset_id and not dataset_id:
|
|
995
|
+
failures.append("registry replay pack manifest does not pin a Future AGI dataset id")
|
|
996
|
+
if active_manifest.registry_version != expected_registry_version:
|
|
997
|
+
failures.append(
|
|
998
|
+
"registry version mismatch: "
|
|
999
|
+
f"manifest {active_manifest.registry_version} != expected {expected_registry_version}"
|
|
1000
|
+
)
|
|
1001
|
+
if not active_manifest.selection_complete:
|
|
1002
|
+
failures.append("registry replay pack selection is incomplete")
|
|
1003
|
+
if active_manifest.coverage_score < min_coverage_score:
|
|
1004
|
+
failures.append(
|
|
1005
|
+
f"coverage score {active_manifest.coverage_score:.4f} below {min_coverage_score:.4f}"
|
|
1006
|
+
)
|
|
1007
|
+
if active_dataset is not None:
|
|
1008
|
+
if loaded_case_count != active_manifest.case_count:
|
|
1009
|
+
failures.append(
|
|
1010
|
+
f"readback case count {loaded_case_count} != manifest {active_manifest.case_count}"
|
|
1011
|
+
)
|
|
1012
|
+
if loaded_case_signature != active_manifest.case_signature:
|
|
1013
|
+
failures.append("readback case signature does not match manifest")
|
|
1014
|
+
if replay_record_count != loaded_case_count:
|
|
1015
|
+
failures.append(
|
|
1016
|
+
f"replay record count {replay_record_count} != readback cases {loaded_case_count}"
|
|
1017
|
+
)
|
|
1018
|
+
if active_optimizer_score is None:
|
|
1019
|
+
failures.append("optimizer replay score is required for registry replay pack promotion")
|
|
1020
|
+
elif active_optimizer_score < min_optimizer_score:
|
|
1021
|
+
failures.append(
|
|
1022
|
+
f"optimizer replay score {active_optimizer_score:.4f} below {min_optimizer_score:.4f}"
|
|
1023
|
+
)
|
|
1024
|
+
|
|
1025
|
+
check_metadata = {
|
|
1026
|
+
"loaded_dataset_metadata": copy.deepcopy(active_dataset.metadata) if active_dataset else {},
|
|
1027
|
+
**dict(metadata or {}),
|
|
1028
|
+
}
|
|
1029
|
+
return AgentRegistryReplayPackPromotionCheck(
|
|
1030
|
+
promotable=not failures,
|
|
1031
|
+
dataset_name=active_manifest.dataset_name,
|
|
1032
|
+
dataset_id=active_dataset_id,
|
|
1033
|
+
registry_version=active_manifest.registry_version,
|
|
1034
|
+
expected_registry_version=expected_registry_version,
|
|
1035
|
+
expected_case_count=active_manifest.case_count,
|
|
1036
|
+
loaded_case_count=loaded_case_count,
|
|
1037
|
+
expected_case_signature=active_manifest.case_signature,
|
|
1038
|
+
loaded_case_signature=loaded_case_signature,
|
|
1039
|
+
coverage_score=active_manifest.coverage_score,
|
|
1040
|
+
min_coverage_score=min_coverage_score,
|
|
1041
|
+
selection_complete=active_manifest.selection_complete,
|
|
1042
|
+
replay_record_count=replay_record_count,
|
|
1043
|
+
optimizer_score=active_optimizer_score,
|
|
1044
|
+
min_optimizer_score=min_optimizer_score,
|
|
1045
|
+
failures=failures,
|
|
1046
|
+
manifest=active_manifest,
|
|
1047
|
+
metadata=check_metadata,
|
|
1048
|
+
)
|
|
1049
|
+
|
|
1050
|
+
|
|
1051
|
+
def compare_futureagi_registry_replay_pack_lineage(
|
|
1052
|
+
manifests: Sequence[AgentRegistryReplayPackManifest | Mapping[str, Any]],
|
|
1053
|
+
*,
|
|
1054
|
+
promotion_checks: Optional[
|
|
1055
|
+
Sequence[AgentRegistryReplayPackPromotionCheck | Mapping[str, Any]]
|
|
1056
|
+
| Mapping[str, AgentRegistryReplayPackPromotionCheck | Mapping[str, Any]]
|
|
1057
|
+
] = None,
|
|
1058
|
+
metadata: Optional[Mapping[str, Any]] = None,
|
|
1059
|
+
) -> AgentRegistryReplayPackLineageReport:
|
|
1060
|
+
"""
|
|
1061
|
+
Compare version-pinned Future AGI registry replay packs.
|
|
1062
|
+
|
|
1063
|
+
This is a local manifest comparison; it does not call Future AGI. Provide
|
|
1064
|
+
promotion checks from `check_futureagi_registry_replay_pack_promotion()` when
|
|
1065
|
+
readback and optimizer replay outcomes should be included in the lineage.
|
|
1066
|
+
"""
|
|
1067
|
+
|
|
1068
|
+
active_manifests = [_coerce_registry_replay_manifest(item) for item in manifests]
|
|
1069
|
+
if not active_manifests:
|
|
1070
|
+
raise ValueError("at least one registry replay-pack manifest is required.")
|
|
1071
|
+
checks_by_key = _registry_replay_promotion_checks_by_key(promotion_checks)
|
|
1072
|
+
entries = [
|
|
1073
|
+
_registry_replay_lineage_entry(
|
|
1074
|
+
manifest,
|
|
1075
|
+
checks_by_key=checks_by_key,
|
|
1076
|
+
)
|
|
1077
|
+
for manifest in active_manifests
|
|
1078
|
+
]
|
|
1079
|
+
transitions = [
|
|
1080
|
+
_registry_replay_lineage_transition(previous, current)
|
|
1081
|
+
for previous, current in zip(entries, entries[1:])
|
|
1082
|
+
]
|
|
1083
|
+
latest = entries[-1]
|
|
1084
|
+
best = max(
|
|
1085
|
+
entries,
|
|
1086
|
+
key=lambda entry: (
|
|
1087
|
+
1 if entry.promotion_promotable else 0,
|
|
1088
|
+
entry.optimizer_score if entry.optimizer_score is not None else float("-inf"),
|
|
1089
|
+
entry.coverage_score,
|
|
1090
|
+
entry.case_count,
|
|
1091
|
+
),
|
|
1092
|
+
)
|
|
1093
|
+
drift_reasons = list(
|
|
1094
|
+
dict.fromkeys(
|
|
1095
|
+
reason
|
|
1096
|
+
for transition in transitions
|
|
1097
|
+
for reason in transition.drift_reasons
|
|
1098
|
+
)
|
|
1099
|
+
)
|
|
1100
|
+
return AgentRegistryReplayPackLineageReport(
|
|
1101
|
+
entry_count=len(entries),
|
|
1102
|
+
entries=entries,
|
|
1103
|
+
transitions=transitions,
|
|
1104
|
+
latest_registry_version=latest.registry_version,
|
|
1105
|
+
latest_dataset_id=latest.dataset_id,
|
|
1106
|
+
latest_promotable=latest.promotion_promotable,
|
|
1107
|
+
best_registry_version=best.registry_version,
|
|
1108
|
+
best_dataset_id=best.dataset_id,
|
|
1109
|
+
best_optimizer_score=best.optimizer_score,
|
|
1110
|
+
drift_detected=bool(drift_reasons),
|
|
1111
|
+
drift_reasons=drift_reasons,
|
|
1112
|
+
metadata={
|
|
1113
|
+
"kind": "futureagi_registry_replay_pack_lineage",
|
|
1114
|
+
**dict(metadata or {}),
|
|
1115
|
+
},
|
|
1116
|
+
)
|
|
1117
|
+
|
|
1118
|
+
|
|
1119
|
+
def triage_futureagi_registry_replay_pack_regression(
|
|
1120
|
+
lineage: AgentRegistryReplayPackLineageReport | Mapping[str, Any],
|
|
1121
|
+
*,
|
|
1122
|
+
max_coverage_drop: float = 0.0,
|
|
1123
|
+
max_optimizer_score_drop: float = 0.02,
|
|
1124
|
+
require_latest_promotable: bool = True,
|
|
1125
|
+
require_readback_match: bool = True,
|
|
1126
|
+
block_on_selected_patch_change: bool = False,
|
|
1127
|
+
block_on_optimizer_backend_change: bool = False,
|
|
1128
|
+
block_on_case_signature_change: bool = False,
|
|
1129
|
+
block_on_retention_key_change: bool = False,
|
|
1130
|
+
block_on_required_contract_removal: bool = True,
|
|
1131
|
+
metadata: Optional[Mapping[str, Any]] = None,
|
|
1132
|
+
) -> AgentRegistryReplayPackTriageReport:
|
|
1133
|
+
"""
|
|
1134
|
+
Recommend whether replay-pack drift should block rollout.
|
|
1135
|
+
|
|
1136
|
+
The triage is local and deterministic. It consumes a lineage report produced
|
|
1137
|
+
by `compare_futureagi_registry_replay_pack_lineage()` and turns coverage,
|
|
1138
|
+
optimizer-score, selected-patch, readback, and promotion drift into an
|
|
1139
|
+
auditable rollout decision.
|
|
1140
|
+
"""
|
|
1141
|
+
|
|
1142
|
+
if max_coverage_drop < 0:
|
|
1143
|
+
raise ValueError("max_coverage_drop must be non-negative.")
|
|
1144
|
+
if max_optimizer_score_drop < 0:
|
|
1145
|
+
raise ValueError("max_optimizer_score_drop must be non-negative.")
|
|
1146
|
+
|
|
1147
|
+
active_lineage = _coerce_registry_replay_lineage_report(lineage)
|
|
1148
|
+
if not active_lineage.entries:
|
|
1149
|
+
raise ValueError("lineage report must contain at least one entry.")
|
|
1150
|
+
|
|
1151
|
+
latest = active_lineage.entries[-1]
|
|
1152
|
+
baseline = active_lineage.entries[-2] if len(active_lineage.entries) > 1 else None
|
|
1153
|
+
latest_transition = active_lineage.transitions[-1] if active_lineage.transitions else None
|
|
1154
|
+
blocking_reasons: list[str] = []
|
|
1155
|
+
warnings: list[str] = []
|
|
1156
|
+
|
|
1157
|
+
if require_latest_promotable:
|
|
1158
|
+
if latest.promotion_promotable is False:
|
|
1159
|
+
blocking_reasons.append("latest_promotion_failed")
|
|
1160
|
+
elif latest.promotion_promotable is None:
|
|
1161
|
+
blocking_reasons.append("missing_latest_promotion_check")
|
|
1162
|
+
if require_readback_match and latest.readback_signature_matches is False:
|
|
1163
|
+
blocking_reasons.append("futureagi_readback_signature_mismatch")
|
|
1164
|
+
if (
|
|
1165
|
+
latest.loaded_case_count is not None
|
|
1166
|
+
and latest.loaded_case_count != latest.case_count
|
|
1167
|
+
):
|
|
1168
|
+
blocking_reasons.append("futureagi_readback_case_count_mismatch")
|
|
1169
|
+
if (
|
|
1170
|
+
latest.replay_record_count is not None
|
|
1171
|
+
and latest.loaded_case_count is not None
|
|
1172
|
+
and latest.replay_record_count != latest.loaded_case_count
|
|
1173
|
+
):
|
|
1174
|
+
blocking_reasons.append("futureagi_replay_record_count_mismatch")
|
|
1175
|
+
if require_latest_promotable and latest.optimizer_score is None:
|
|
1176
|
+
blocking_reasons.append("missing_latest_optimizer_score")
|
|
1177
|
+
if latest.failures and latest.promotion_promotable is not True:
|
|
1178
|
+
blocking_reasons.append("latest_promotion_failures")
|
|
1179
|
+
|
|
1180
|
+
coverage_delta: Optional[float] = None
|
|
1181
|
+
optimizer_score_delta: Optional[float] = None
|
|
1182
|
+
if latest_transition is not None:
|
|
1183
|
+
coverage_delta = latest_transition.coverage_delta
|
|
1184
|
+
optimizer_score_delta = latest_transition.optimizer_score_delta
|
|
1185
|
+
if latest_transition.coverage_delta < -max_coverage_drop:
|
|
1186
|
+
blocking_reasons.append("coverage_regression")
|
|
1187
|
+
elif latest_transition.coverage_delta != 0:
|
|
1188
|
+
warnings.append("coverage_drift")
|
|
1189
|
+
if latest_transition.optimizer_score_delta is None:
|
|
1190
|
+
if latest.optimizer_score is None or (baseline and baseline.optimizer_score is None):
|
|
1191
|
+
warnings.append("missing_optimizer_score_delta")
|
|
1192
|
+
elif latest_transition.optimizer_score_delta < -max_optimizer_score_drop:
|
|
1193
|
+
blocking_reasons.append("optimizer_score_regression")
|
|
1194
|
+
elif latest_transition.optimizer_score_delta < 0:
|
|
1195
|
+
warnings.append("optimizer_score_drift")
|
|
1196
|
+
if latest_transition.selected_patch_changed:
|
|
1197
|
+
if block_on_selected_patch_change:
|
|
1198
|
+
blocking_reasons.append("selected_patch_changed")
|
|
1199
|
+
else:
|
|
1200
|
+
warnings.append("selected_patch_changed")
|
|
1201
|
+
if latest_transition.optimizer_backend_changed:
|
|
1202
|
+
if block_on_optimizer_backend_change:
|
|
1203
|
+
blocking_reasons.append("optimizer_backend_changed")
|
|
1204
|
+
else:
|
|
1205
|
+
warnings.append("optimizer_backend_changed")
|
|
1206
|
+
if latest_transition.case_signature_changed:
|
|
1207
|
+
if block_on_case_signature_change:
|
|
1208
|
+
blocking_reasons.append("case_signature_changed")
|
|
1209
|
+
else:
|
|
1210
|
+
warnings.append("case_signature_changed")
|
|
1211
|
+
if latest_transition.retention_key_changed:
|
|
1212
|
+
if block_on_retention_key_change:
|
|
1213
|
+
blocking_reasons.append("retention_key_changed")
|
|
1214
|
+
else:
|
|
1215
|
+
warnings.append("retention_key_changed")
|
|
1216
|
+
if latest_transition.promotion_status_changed:
|
|
1217
|
+
warnings.append("promotion_status_changed")
|
|
1218
|
+
if latest_transition.removed_required_presets:
|
|
1219
|
+
if block_on_required_contract_removal:
|
|
1220
|
+
blocking_reasons.append("required_presets_removed")
|
|
1221
|
+
else:
|
|
1222
|
+
warnings.append("required_presets_removed")
|
|
1223
|
+
if latest_transition.removed_invariant_families:
|
|
1224
|
+
if block_on_required_contract_removal:
|
|
1225
|
+
blocking_reasons.append("required_invariant_families_removed")
|
|
1226
|
+
else:
|
|
1227
|
+
warnings.append("required_invariant_families_removed")
|
|
1228
|
+
if (
|
|
1229
|
+
latest_transition.added_required_presets
|
|
1230
|
+
or latest_transition.added_invariant_families
|
|
1231
|
+
):
|
|
1232
|
+
warnings.append("required_contract_expanded")
|
|
1233
|
+
|
|
1234
|
+
best_optimizer_score_gap: Optional[float] = None
|
|
1235
|
+
if (
|
|
1236
|
+
latest.optimizer_score is not None
|
|
1237
|
+
and active_lineage.best_optimizer_score is not None
|
|
1238
|
+
):
|
|
1239
|
+
best_optimizer_score_gap = round(
|
|
1240
|
+
latest.optimizer_score - active_lineage.best_optimizer_score,
|
|
1241
|
+
8,
|
|
1242
|
+
)
|
|
1243
|
+
if best_optimizer_score_gap < -max_optimizer_score_drop:
|
|
1244
|
+
blocking_reasons.append("latest_below_best_optimizer_score")
|
|
1245
|
+
|
|
1246
|
+
blocking_reasons = _unique_strings(blocking_reasons)
|
|
1247
|
+
warnings = [
|
|
1248
|
+
warning
|
|
1249
|
+
for warning in _unique_strings(warnings)
|
|
1250
|
+
if warning not in blocking_reasons
|
|
1251
|
+
]
|
|
1252
|
+
decision = "block" if blocking_reasons else ("review" if warnings else "promote")
|
|
1253
|
+
severity = _registry_replay_triage_severity(
|
|
1254
|
+
blocking_reasons=blocking_reasons,
|
|
1255
|
+
warnings=warnings,
|
|
1256
|
+
)
|
|
1257
|
+
recommendations = _registry_replay_triage_recommendations(
|
|
1258
|
+
blocking_reasons=blocking_reasons,
|
|
1259
|
+
warnings=warnings,
|
|
1260
|
+
)
|
|
1261
|
+
return AgentRegistryReplayPackTriageReport(
|
|
1262
|
+
decision=decision,
|
|
1263
|
+
severity=severity,
|
|
1264
|
+
block_rollout=bool(blocking_reasons),
|
|
1265
|
+
latest_registry_version=latest.registry_version,
|
|
1266
|
+
latest_dataset_id=latest.dataset_id,
|
|
1267
|
+
baseline_registry_version=baseline.registry_version if baseline else None,
|
|
1268
|
+
baseline_dataset_id=baseline.dataset_id if baseline else None,
|
|
1269
|
+
best_registry_version=active_lineage.best_registry_version,
|
|
1270
|
+
best_dataset_id=active_lineage.best_dataset_id,
|
|
1271
|
+
latest_promotable=latest.promotion_promotable,
|
|
1272
|
+
coverage_delta=coverage_delta,
|
|
1273
|
+
optimizer_score_delta=optimizer_score_delta,
|
|
1274
|
+
best_optimizer_score_gap=best_optimizer_score_gap,
|
|
1275
|
+
blocking_reasons=blocking_reasons,
|
|
1276
|
+
warnings=warnings,
|
|
1277
|
+
drift_reasons=list(active_lineage.drift_reasons),
|
|
1278
|
+
recommendations=recommendations,
|
|
1279
|
+
evidence={
|
|
1280
|
+
"thresholds": {
|
|
1281
|
+
"max_coverage_drop": max_coverage_drop,
|
|
1282
|
+
"max_optimizer_score_drop": max_optimizer_score_drop,
|
|
1283
|
+
"require_latest_promotable": require_latest_promotable,
|
|
1284
|
+
"require_readback_match": require_readback_match,
|
|
1285
|
+
"block_on_selected_patch_change": block_on_selected_patch_change,
|
|
1286
|
+
"block_on_optimizer_backend_change": block_on_optimizer_backend_change,
|
|
1287
|
+
"block_on_case_signature_change": block_on_case_signature_change,
|
|
1288
|
+
"block_on_retention_key_change": block_on_retention_key_change,
|
|
1289
|
+
"block_on_required_contract_removal": block_on_required_contract_removal,
|
|
1290
|
+
},
|
|
1291
|
+
"latest_entry": latest.model_dump(),
|
|
1292
|
+
"baseline_entry": baseline.model_dump() if baseline else None,
|
|
1293
|
+
"latest_transition": (
|
|
1294
|
+
latest_transition.model_dump() if latest_transition else None
|
|
1295
|
+
),
|
|
1296
|
+
"latest_failures": list(latest.failures),
|
|
1297
|
+
},
|
|
1298
|
+
metadata={
|
|
1299
|
+
"kind": "futureagi_registry_replay_pack_regression_triage",
|
|
1300
|
+
**dict(metadata or {}),
|
|
1301
|
+
},
|
|
1302
|
+
)
|
|
1303
|
+
|
|
1304
|
+
|
|
1305
|
+
def load_futureagi_experiment_history(
|
|
1306
|
+
*,
|
|
1307
|
+
experiment_id: str,
|
|
1308
|
+
candidate: Optional[AgentCandidate] = None,
|
|
1309
|
+
required_metrics: Optional[Mapping[str, float]] = None,
|
|
1310
|
+
required_trace_signals: Optional[Sequence[str]] = None,
|
|
1311
|
+
fi_api_key: Optional[str] = None,
|
|
1312
|
+
fi_secret_key: Optional[str] = None,
|
|
1313
|
+
fi_base_url: Optional[str] = None,
|
|
1314
|
+
client: Any = None,
|
|
1315
|
+
page_size: int = 100,
|
|
1316
|
+
max_pages: int = 20,
|
|
1317
|
+
timeout: float = 30.0,
|
|
1318
|
+
include_rows: bool = True,
|
|
1319
|
+
include_stats: bool = True,
|
|
1320
|
+
prefer_v2: bool = True,
|
|
1321
|
+
metadata: Optional[Mapping[str, Any]] = None,
|
|
1322
|
+
) -> AgentObservabilityWindow:
|
|
1323
|
+
"""
|
|
1324
|
+
Pull Future AGI experiment history into optimizer feedback.
|
|
1325
|
+
|
|
1326
|
+
Unlike `load_futureagi_regression_dataset()`, this reads native Future AGI
|
|
1327
|
+
experiment detail/stats/row payloads instead of regression-pack rows. The
|
|
1328
|
+
resulting observability window can drive `AgentFeedbackOptimizer` through
|
|
1329
|
+
any metric-bound backend: deterministic, council, society, social-memory,
|
|
1330
|
+
curriculum, evolutionary, TPE, Pareto, or bandit.
|
|
1331
|
+
"""
|
|
1332
|
+
|
|
1333
|
+
if not experiment_id:
|
|
1334
|
+
raise ValueError("load_futureagi_experiment_history requires experiment_id.")
|
|
1335
|
+
if page_size < 1:
|
|
1336
|
+
raise ValueError("page_size must be at least 1.")
|
|
1337
|
+
if max_pages < 1:
|
|
1338
|
+
raise ValueError("max_pages must be at least 1.")
|
|
1339
|
+
|
|
1340
|
+
active_endpoint = (fi_base_url or os.getenv("FI_BASE_URL") or "https://api.futureagi.com").rstrip("/")
|
|
1341
|
+
if client is None:
|
|
1342
|
+
active_api_key = fi_api_key or os.getenv("FI_API_KEY")
|
|
1343
|
+
active_secret_key = fi_secret_key or os.getenv("FI_SECRET_KEY")
|
|
1344
|
+
if not active_api_key or not active_secret_key:
|
|
1345
|
+
raise ValueError(
|
|
1346
|
+
"Future AGI experiment-history loading requires FI_API_KEY and "
|
|
1347
|
+
"FI_SECRET_KEY or an injected Future AGI experiment client."
|
|
1348
|
+
)
|
|
1349
|
+
client = _load_futureagi_experiment_reader_client(
|
|
1350
|
+
fi_api_key=active_api_key,
|
|
1351
|
+
fi_secret_key=active_secret_key,
|
|
1352
|
+
fi_base_url=active_endpoint,
|
|
1353
|
+
timeout=timeout,
|
|
1354
|
+
)
|
|
1355
|
+
if client is None:
|
|
1356
|
+
raise RuntimeError(
|
|
1357
|
+
"Future AGI experiment-history loading requires the "
|
|
1358
|
+
"`ai-evaluation` HTTP auth primitives."
|
|
1359
|
+
)
|
|
1360
|
+
|
|
1361
|
+
payload = _futureagi_experiment_payload(
|
|
1362
|
+
client,
|
|
1363
|
+
experiment_id=experiment_id,
|
|
1364
|
+
page_size=page_size,
|
|
1365
|
+
max_pages=max_pages,
|
|
1366
|
+
include_rows=include_rows,
|
|
1367
|
+
include_stats=include_stats,
|
|
1368
|
+
prefer_v2=prefer_v2,
|
|
1369
|
+
)
|
|
1370
|
+
thresholds = {
|
|
1371
|
+
str(key): float(value)
|
|
1372
|
+
for key, value in dict(required_metrics or {}).items()
|
|
1373
|
+
}
|
|
1374
|
+
required_signals = [_normalize_signal(item) for item in required_trace_signals or []]
|
|
1375
|
+
required_signals = [item for item in required_signals if item]
|
|
1376
|
+
experiment_metadata = _futureagi_experiment_metadata(payload, experiment_id=experiment_id)
|
|
1377
|
+
raw_records = _futureagi_experiment_observation_records(
|
|
1378
|
+
payload,
|
|
1379
|
+
experiment_id=experiment_id,
|
|
1380
|
+
experiment_metadata=experiment_metadata,
|
|
1381
|
+
)
|
|
1382
|
+
records = [
|
|
1383
|
+
_normalize_observability_record(
|
|
1384
|
+
raw_record,
|
|
1385
|
+
index=index,
|
|
1386
|
+
candidate=candidate,
|
|
1387
|
+
source="futureagi",
|
|
1388
|
+
framework=str(
|
|
1389
|
+
experiment_metadata.get("framework")
|
|
1390
|
+
or experiment_metadata.get("runtime")
|
|
1391
|
+
or "generic"
|
|
1392
|
+
),
|
|
1393
|
+
required_metrics=thresholds,
|
|
1394
|
+
required_trace_signals=required_signals,
|
|
1395
|
+
)
|
|
1396
|
+
for index, raw_record in enumerate(raw_records, start=1)
|
|
1397
|
+
]
|
|
1398
|
+
_penalize_missing_futureagi_experiment_metrics(records, thresholds)
|
|
1399
|
+
|
|
1400
|
+
window_metadata = {
|
|
1401
|
+
"kind": "futureagi_experiment_history",
|
|
1402
|
+
"schema_version": FUTUREAGI_EXPERIMENT_HISTORY_SCHEMA_VERSION,
|
|
1403
|
+
"experiment_id": experiment_id,
|
|
1404
|
+
"experiment_name": experiment_metadata.get("name"),
|
|
1405
|
+
"experiment_status": experiment_metadata.get("status"),
|
|
1406
|
+
"endpoint": active_endpoint,
|
|
1407
|
+
"record_count": len(records),
|
|
1408
|
+
"payload_sections": sorted(str(key) for key in payload.keys()),
|
|
1409
|
+
**dict(metadata or {}),
|
|
1410
|
+
}
|
|
1411
|
+
return AgentObservabilityWindow(
|
|
1412
|
+
source=_resolve_window_source(records, fallback="futureagi"),
|
|
1413
|
+
framework=_resolve_window_framework(records, fallback=str(window_metadata.get("framework") or "generic")),
|
|
1414
|
+
candidate=candidate,
|
|
1415
|
+
records=records,
|
|
1416
|
+
required_metrics=thresholds,
|
|
1417
|
+
required_trace_signals=required_signals,
|
|
1418
|
+
metadata=window_metadata,
|
|
1419
|
+
)
|
|
1420
|
+
|
|
1421
|
+
|
|
1422
|
+
def build_agent_regression_dataset(
|
|
1423
|
+
windows: AgentObservabilityWindow | Sequence[AgentObservabilityWindow],
|
|
1424
|
+
*,
|
|
1425
|
+
name: str = "observability-regression",
|
|
1426
|
+
failed_only: bool = True,
|
|
1427
|
+
include_raw: bool = True,
|
|
1428
|
+
metadata: Optional[Mapping[str, Any]] = None,
|
|
1429
|
+
) -> AgentRegressionDataset:
|
|
1430
|
+
"""
|
|
1431
|
+
Convert normalized observability windows into durable regression cases.
|
|
1432
|
+
|
|
1433
|
+
The resulting cases preserve the production signal that failed, the expected
|
|
1434
|
+
metric/trace thresholds a repaired candidate must satisfy, and export
|
|
1435
|
+
helpers for local replay plus Future AGI datasets.
|
|
1436
|
+
"""
|
|
1437
|
+
|
|
1438
|
+
normalized_windows = _regression_windows(windows)
|
|
1439
|
+
cases: list[AgentRegressionCase] = []
|
|
1440
|
+
for window_index, window in enumerate(normalized_windows, start=1):
|
|
1441
|
+
for record in window.records:
|
|
1442
|
+
if failed_only and record.passed:
|
|
1443
|
+
continue
|
|
1444
|
+
cases.append(
|
|
1445
|
+
_regression_case_from_observability_record(
|
|
1446
|
+
record,
|
|
1447
|
+
window=window,
|
|
1448
|
+
window_index=window_index,
|
|
1449
|
+
include_raw=include_raw,
|
|
1450
|
+
)
|
|
1451
|
+
)
|
|
1452
|
+
|
|
1453
|
+
return AgentRegressionDataset(
|
|
1454
|
+
name=name,
|
|
1455
|
+
source=_regression_source(normalized_windows),
|
|
1456
|
+
framework=_regression_framework(normalized_windows),
|
|
1457
|
+
cases=cases,
|
|
1458
|
+
metadata={
|
|
1459
|
+
"kind": "observability_regression_dataset",
|
|
1460
|
+
"failed_only": failed_only,
|
|
1461
|
+
"include_raw": include_raw,
|
|
1462
|
+
"window_count": len(normalized_windows),
|
|
1463
|
+
"record_count": sum(len(window.records) for window in normalized_windows),
|
|
1464
|
+
"case_count": len(cases),
|
|
1465
|
+
**dict(metadata or {}),
|
|
1466
|
+
},
|
|
1467
|
+
)
|
|
1468
|
+
|
|
1469
|
+
|
|
1470
|
+
def build_agent_regression_dataset_coverage_report(
|
|
1471
|
+
dataset: AgentRegressionDataset,
|
|
1472
|
+
*,
|
|
1473
|
+
target: Any = None,
|
|
1474
|
+
metric_path_hints: Optional[Mapping[str, Sequence[str]]] = None,
|
|
1475
|
+
tag_path_hints: Optional[Mapping[str, Sequence[str]]] = None,
|
|
1476
|
+
metadata: Optional[Mapping[str, Any]] = None,
|
|
1477
|
+
) -> AgentRegressionDatasetCoverageReport:
|
|
1478
|
+
"""Summarize replay-pack coverage before Future AGI publish or optimizer replay."""
|
|
1479
|
+
|
|
1480
|
+
target_paths = _coverage_target_paths(target)
|
|
1481
|
+
target_path_set = set(target_paths)
|
|
1482
|
+
metric_hints = _coverage_path_hints(metric_path_hints)
|
|
1483
|
+
tag_hints = _coverage_path_hints(tag_path_hints)
|
|
1484
|
+
source_counts: dict[str, int] = {}
|
|
1485
|
+
framework_counts: dict[str, int] = {}
|
|
1486
|
+
tag_counts: dict[str, int] = {}
|
|
1487
|
+
observed_metric_case_counts: dict[str, int] = {}
|
|
1488
|
+
required_metric_case_counts: dict[str, int] = {}
|
|
1489
|
+
failed_metric_case_counts: dict[str, int] = {}
|
|
1490
|
+
required_metrics: dict[str, float] = {}
|
|
1491
|
+
required_trace_signals: list[str] = []
|
|
1492
|
+
seen_required_signals: set[str] = set()
|
|
1493
|
+
trace_signal_case_counts: dict[str, int] = {}
|
|
1494
|
+
missing_trace_signal_case_counts: dict[str, int] = {}
|
|
1495
|
+
search_path_case_counts: dict[str, int] = {path: 0 for path in target_paths}
|
|
1496
|
+
failure_examples: dict[str, list[str]] = {}
|
|
1497
|
+
failed_case_count = 0
|
|
1498
|
+
passed_case_count = 0
|
|
1499
|
+
|
|
1500
|
+
for case in dataset.cases:
|
|
1501
|
+
observability = _ensure_mapping(case.input.get("observability"))
|
|
1502
|
+
expected = _ensure_mapping(case.expected)
|
|
1503
|
+
source = str(observability.get("source") or case.metadata.get("source") or dataset.source)
|
|
1504
|
+
framework = str(
|
|
1505
|
+
observability.get("framework")
|
|
1506
|
+
or case.metadata.get("framework")
|
|
1507
|
+
or dataset.framework
|
|
1508
|
+
)
|
|
1509
|
+
_count(source_counts, source)
|
|
1510
|
+
_count(framework_counts, framework)
|
|
1511
|
+
|
|
1512
|
+
passed_value = observability.get("passed")
|
|
1513
|
+
passed = bool(passed_value) if isinstance(passed_value, bool) else bool(case.metadata.get("passed"))
|
|
1514
|
+
if passed:
|
|
1515
|
+
passed_case_count += 1
|
|
1516
|
+
else:
|
|
1517
|
+
failed_case_count += 1
|
|
1518
|
+
|
|
1519
|
+
tags = [str(tag) for tag in case.tags]
|
|
1520
|
+
for tag in tags:
|
|
1521
|
+
_count(tag_counts, tag)
|
|
1522
|
+
|
|
1523
|
+
metrics = _float_mapping(observability.get("metrics"))
|
|
1524
|
+
for metric in sorted(metrics):
|
|
1525
|
+
_count(observed_metric_case_counts, metric)
|
|
1526
|
+
|
|
1527
|
+
case_required_metrics = _float_mapping(expected.get("required_metrics"))
|
|
1528
|
+
for metric, threshold in sorted(case_required_metrics.items()):
|
|
1529
|
+
_count(required_metric_case_counts, metric)
|
|
1530
|
+
required_metrics[metric] = max(threshold, required_metrics.get(metric, threshold))
|
|
1531
|
+
observed = metrics.get(metric)
|
|
1532
|
+
if observed is None or observed < threshold:
|
|
1533
|
+
_count(failed_metric_case_counts, metric)
|
|
1534
|
+
|
|
1535
|
+
trace_signals = {
|
|
1536
|
+
_normalize_signal(signal)
|
|
1537
|
+
for signal in _string_list(observability.get("trace_signals"))
|
|
1538
|
+
if _normalize_signal(signal)
|
|
1539
|
+
}
|
|
1540
|
+
for signal in sorted(trace_signals):
|
|
1541
|
+
_count(trace_signal_case_counts, signal)
|
|
1542
|
+
|
|
1543
|
+
for signal in _string_list(expected.get("required_trace_signals")):
|
|
1544
|
+
normalized = _normalize_signal(signal)
|
|
1545
|
+
if not normalized:
|
|
1546
|
+
continue
|
|
1547
|
+
if normalized not in seen_required_signals:
|
|
1548
|
+
required_trace_signals.append(normalized)
|
|
1549
|
+
seen_required_signals.add(normalized)
|
|
1550
|
+
if normalized not in trace_signals:
|
|
1551
|
+
_count(missing_trace_signal_case_counts, normalized)
|
|
1552
|
+
|
|
1553
|
+
failures = _string_list(observability.get("failures") or expected.get("previous_failures"))
|
|
1554
|
+
for failure in failures[:8]:
|
|
1555
|
+
family = _coverage_failure_family(failure)
|
|
1556
|
+
examples = failure_examples.setdefault(family, [])
|
|
1557
|
+
if case.id not in examples and len(examples) < 5:
|
|
1558
|
+
examples.append(case.id)
|
|
1559
|
+
|
|
1560
|
+
for path in _coverage_search_path_hits(
|
|
1561
|
+
tags=tags,
|
|
1562
|
+
metrics=[*metrics.keys(), *case_required_metrics.keys()],
|
|
1563
|
+
failures=failures,
|
|
1564
|
+
target_paths=target_paths,
|
|
1565
|
+
target_path_set=target_path_set,
|
|
1566
|
+
metric_path_hints=metric_hints,
|
|
1567
|
+
tag_path_hints=tag_hints,
|
|
1568
|
+
):
|
|
1569
|
+
search_path_case_counts[path] = search_path_case_counts.get(path, 0) + 1
|
|
1570
|
+
|
|
1571
|
+
all_required_metrics = sorted(required_metrics)
|
|
1572
|
+
uncovered_required_metrics = [
|
|
1573
|
+
metric
|
|
1574
|
+
for metric in all_required_metrics
|
|
1575
|
+
if required_metric_case_counts.get(metric, 0) == 0
|
|
1576
|
+
]
|
|
1577
|
+
uncovered_search_paths = [
|
|
1578
|
+
path for path in target_paths if search_path_case_counts.get(path, 0) == 0
|
|
1579
|
+
]
|
|
1580
|
+
|
|
1581
|
+
return AgentRegressionDatasetCoverageReport(
|
|
1582
|
+
dataset_name=dataset.name,
|
|
1583
|
+
source=dataset.source,
|
|
1584
|
+
framework=dataset.framework,
|
|
1585
|
+
case_count=len(dataset.cases),
|
|
1586
|
+
failed_case_count=failed_case_count,
|
|
1587
|
+
passed_case_count=passed_case_count,
|
|
1588
|
+
source_counts=dict(sorted(source_counts.items())),
|
|
1589
|
+
framework_counts=dict(sorted(framework_counts.items())),
|
|
1590
|
+
tag_counts=dict(sorted(tag_counts.items())),
|
|
1591
|
+
observed_metric_case_counts=dict(sorted(observed_metric_case_counts.items())),
|
|
1592
|
+
required_metric_case_counts=dict(sorted(required_metric_case_counts.items())),
|
|
1593
|
+
failed_metric_case_counts=dict(sorted(failed_metric_case_counts.items())),
|
|
1594
|
+
required_metrics={key: required_metrics[key] for key in all_required_metrics},
|
|
1595
|
+
required_trace_signals=required_trace_signals,
|
|
1596
|
+
trace_signal_case_counts=dict(sorted(trace_signal_case_counts.items())),
|
|
1597
|
+
missing_trace_signal_case_counts=dict(
|
|
1598
|
+
sorted(missing_trace_signal_case_counts.items())
|
|
1599
|
+
),
|
|
1600
|
+
search_path_case_counts={
|
|
1601
|
+
key: search_path_case_counts[key]
|
|
1602
|
+
for key in [*target_paths, *sorted(path for path in search_path_case_counts if path not in target_path_set)]
|
|
1603
|
+
},
|
|
1604
|
+
uncovered_required_metrics=uncovered_required_metrics,
|
|
1605
|
+
uncovered_search_paths=uncovered_search_paths,
|
|
1606
|
+
failure_examples=dict(sorted(failure_examples.items())),
|
|
1607
|
+
metadata={
|
|
1608
|
+
"kind": "regression_dataset_coverage_report",
|
|
1609
|
+
"target_name": getattr(target, "name", None),
|
|
1610
|
+
"target_layers": list(getattr(target, "layers", []) or []),
|
|
1611
|
+
"metric_path_hints": {
|
|
1612
|
+
key: list(value) for key, value in metric_hints.items()
|
|
1613
|
+
},
|
|
1614
|
+
"tag_path_hints": {key: list(value) for key, value in tag_hints.items()},
|
|
1615
|
+
**dict(metadata or {}),
|
|
1616
|
+
},
|
|
1617
|
+
)
|
|
1618
|
+
|
|
1619
|
+
|
|
1620
|
+
def _count(counts: dict[str, int], key: str) -> None:
|
|
1621
|
+
counts[str(key)] = counts.get(str(key), 0) + 1
|
|
1622
|
+
|
|
1623
|
+
|
|
1624
|
+
def _coverage_target_paths(target: Any) -> list[str]:
|
|
1625
|
+
search_space = getattr(target, "search_space", None)
|
|
1626
|
+
if isinstance(search_space, Mapping):
|
|
1627
|
+
return [str(path) for path in search_space]
|
|
1628
|
+
return []
|
|
1629
|
+
|
|
1630
|
+
|
|
1631
|
+
def _coverage_path_hints(
|
|
1632
|
+
value: Optional[Mapping[str, Sequence[str]]],
|
|
1633
|
+
) -> dict[str, list[str]]:
|
|
1634
|
+
hints: dict[str, list[str]] = {}
|
|
1635
|
+
for key, paths in dict(value or {}).items():
|
|
1636
|
+
hints[str(key)] = [str(path) for path in paths]
|
|
1637
|
+
return hints
|
|
1638
|
+
|
|
1639
|
+
|
|
1640
|
+
def _coverage_search_path_hits(
|
|
1641
|
+
*,
|
|
1642
|
+
tags: Sequence[str],
|
|
1643
|
+
metrics: Sequence[str],
|
|
1644
|
+
failures: Sequence[str],
|
|
1645
|
+
target_paths: Sequence[str],
|
|
1646
|
+
target_path_set: set[str],
|
|
1647
|
+
metric_path_hints: Mapping[str, Sequence[str]],
|
|
1648
|
+
tag_path_hints: Mapping[str, Sequence[str]],
|
|
1649
|
+
) -> list[str]:
|
|
1650
|
+
hits: set[str] = set()
|
|
1651
|
+
for metric in metrics:
|
|
1652
|
+
for path in metric_path_hints.get(str(metric), ()):
|
|
1653
|
+
if not target_path_set or path in target_path_set:
|
|
1654
|
+
hits.add(path)
|
|
1655
|
+
for tag in tags:
|
|
1656
|
+
for path in tag_path_hints.get(str(tag), ()):
|
|
1657
|
+
if not target_path_set or path in target_path_set:
|
|
1658
|
+
hits.add(path)
|
|
1659
|
+
|
|
1660
|
+
if target_paths:
|
|
1661
|
+
text = " ".join([*tags, *metrics, *failures]).lower()
|
|
1662
|
+
text_tokens = set(_case_slug(text).split("-"))
|
|
1663
|
+
for path in target_paths:
|
|
1664
|
+
path_parts_list = [part.lower() for part in path.split(".") if part]
|
|
1665
|
+
matching_parts = (
|
|
1666
|
+
path_parts_list[1:] if len(path_parts_list) > 1 else path_parts_list
|
|
1667
|
+
)
|
|
1668
|
+
path_tokens = set(_case_slug(".".join(matching_parts)).split("-"))
|
|
1669
|
+
path_parts = set(matching_parts)
|
|
1670
|
+
if text_tokens.intersection(path_tokens) or text_tokens.intersection(path_parts):
|
|
1671
|
+
hits.add(path)
|
|
1672
|
+
|
|
1673
|
+
return [path for path in target_paths if path in hits] + sorted(
|
|
1674
|
+
path for path in hits if path not in target_path_set
|
|
1675
|
+
)
|
|
1676
|
+
|
|
1677
|
+
|
|
1678
|
+
def _coverage_failure_family(failure: Any) -> str:
|
|
1679
|
+
text = str(failure or "unknown")
|
|
1680
|
+
metric_match = _METRIC_FAILURE_RE.search(text)
|
|
1681
|
+
if metric_match:
|
|
1682
|
+
return f"metric:{metric_match.group(1)}"
|
|
1683
|
+
slug = _case_slug(text)
|
|
1684
|
+
return slug or "unknown"
|
|
1685
|
+
|
|
1686
|
+
|
|
1687
|
+
_METRIC_FAILURE_RE = re.compile(r"metric '([^']+)'")
|
|
1688
|
+
|
|
1689
|
+
|
|
1690
|
+
def _futureagi_dataset_columns() -> list[dict[str, str]]:
|
|
1691
|
+
return [dict(column) for column in FUTUREAGI_REGRESSION_DATASET_COLUMNS]
|
|
1692
|
+
|
|
1693
|
+
|
|
1694
|
+
def _load_futureagi_dataset_client(
|
|
1695
|
+
*,
|
|
1696
|
+
fi_api_key: str,
|
|
1697
|
+
fi_secret_key: str,
|
|
1698
|
+
fi_base_url: str,
|
|
1699
|
+
timeout: float,
|
|
1700
|
+
) -> Any:
|
|
1701
|
+
try:
|
|
1702
|
+
from fi.datasets import Dataset # type: ignore
|
|
1703
|
+
from fi.datasets.types import ( # type: ignore
|
|
1704
|
+
Cell,
|
|
1705
|
+
Column,
|
|
1706
|
+
DatasetConfig,
|
|
1707
|
+
DataTypeChoices,
|
|
1708
|
+
ModelTypes,
|
|
1709
|
+
Row,
|
|
1710
|
+
SourceChoices,
|
|
1711
|
+
)
|
|
1712
|
+
|
|
1713
|
+
return _FutureAGISDKDatasetPublisher(
|
|
1714
|
+
Dataset=Dataset,
|
|
1715
|
+
DatasetConfig=DatasetConfig,
|
|
1716
|
+
Column=Column,
|
|
1717
|
+
Row=Row,
|
|
1718
|
+
Cell=Cell,
|
|
1719
|
+
DataTypeChoices=DataTypeChoices,
|
|
1720
|
+
ModelTypes=ModelTypes,
|
|
1721
|
+
SourceChoices=SourceChoices,
|
|
1722
|
+
fi_api_key=fi_api_key,
|
|
1723
|
+
fi_secret_key=fi_secret_key,
|
|
1724
|
+
fi_base_url=fi_base_url,
|
|
1725
|
+
timeout=timeout,
|
|
1726
|
+
)
|
|
1727
|
+
except Exception:
|
|
1728
|
+
pass
|
|
1729
|
+
|
|
1730
|
+
try:
|
|
1731
|
+
from fi.api.auth import APIKeyAuth # type: ignore
|
|
1732
|
+
from fi.api.types import HttpMethod, RequestConfig # type: ignore
|
|
1733
|
+
from fi.utils.routes import Routes # type: ignore
|
|
1734
|
+
except Exception:
|
|
1735
|
+
return None
|
|
1736
|
+
|
|
1737
|
+
return _FutureAGIHttpDatasetPublisher(
|
|
1738
|
+
APIKeyAuth=APIKeyAuth,
|
|
1739
|
+
HttpMethod=HttpMethod,
|
|
1740
|
+
RequestConfig=RequestConfig,
|
|
1741
|
+
Routes=Routes,
|
|
1742
|
+
fi_api_key=fi_api_key,
|
|
1743
|
+
fi_secret_key=fi_secret_key,
|
|
1744
|
+
fi_base_url=fi_base_url,
|
|
1745
|
+
timeout=timeout,
|
|
1746
|
+
)
|
|
1747
|
+
|
|
1748
|
+
|
|
1749
|
+
def _load_futureagi_dataset_reader_client(
|
|
1750
|
+
*,
|
|
1751
|
+
fi_api_key: str,
|
|
1752
|
+
fi_secret_key: str,
|
|
1753
|
+
fi_base_url: str,
|
|
1754
|
+
timeout: float,
|
|
1755
|
+
) -> Any:
|
|
1756
|
+
try:
|
|
1757
|
+
from fi.api.auth import APIKeyAuth # type: ignore
|
|
1758
|
+
from fi.api.types import HttpMethod, RequestConfig # type: ignore
|
|
1759
|
+
from fi.utils.routes import Routes # type: ignore
|
|
1760
|
+
except Exception:
|
|
1761
|
+
return None
|
|
1762
|
+
|
|
1763
|
+
return _FutureAGIHttpDatasetReader(
|
|
1764
|
+
APIKeyAuth=APIKeyAuth,
|
|
1765
|
+
HttpMethod=HttpMethod,
|
|
1766
|
+
RequestConfig=RequestConfig,
|
|
1767
|
+
Routes=Routes,
|
|
1768
|
+
fi_api_key=fi_api_key,
|
|
1769
|
+
fi_secret_key=fi_secret_key,
|
|
1770
|
+
fi_base_url=fi_base_url,
|
|
1771
|
+
timeout=timeout,
|
|
1772
|
+
)
|
|
1773
|
+
|
|
1774
|
+
|
|
1775
|
+
def _load_futureagi_experiment_reader_client(
|
|
1776
|
+
*,
|
|
1777
|
+
fi_api_key: str,
|
|
1778
|
+
fi_secret_key: str,
|
|
1779
|
+
fi_base_url: str,
|
|
1780
|
+
timeout: float,
|
|
1781
|
+
) -> Any:
|
|
1782
|
+
try:
|
|
1783
|
+
from fi.api.auth import APIKeyAuth # type: ignore
|
|
1784
|
+
from fi.api.types import HttpMethod, RequestConfig # type: ignore
|
|
1785
|
+
except Exception:
|
|
1786
|
+
return None
|
|
1787
|
+
|
|
1788
|
+
return _FutureAGIHttpExperimentReader(
|
|
1789
|
+
APIKeyAuth=APIKeyAuth,
|
|
1790
|
+
HttpMethod=HttpMethod,
|
|
1791
|
+
RequestConfig=RequestConfig,
|
|
1792
|
+
fi_api_key=fi_api_key,
|
|
1793
|
+
fi_secret_key=fi_secret_key,
|
|
1794
|
+
fi_base_url=fi_base_url,
|
|
1795
|
+
timeout=timeout,
|
|
1796
|
+
)
|
|
1797
|
+
|
|
1798
|
+
|
|
1799
|
+
def _publish_futureagi_regression_dataset_with_client(
|
|
1800
|
+
client: Any,
|
|
1801
|
+
*,
|
|
1802
|
+
dataset_name: str,
|
|
1803
|
+
dataset_id: Optional[str],
|
|
1804
|
+
columns: Sequence[Mapping[str, Any]],
|
|
1805
|
+
rows: Sequence[Mapping[str, Any]],
|
|
1806
|
+
metadata: Mapping[str, Any],
|
|
1807
|
+
) -> Any:
|
|
1808
|
+
for method_name in (
|
|
1809
|
+
"publish_regression_dataset",
|
|
1810
|
+
"create_regression_dataset",
|
|
1811
|
+
"create_dataset",
|
|
1812
|
+
):
|
|
1813
|
+
method = getattr(client, method_name, None)
|
|
1814
|
+
if not callable(method):
|
|
1815
|
+
continue
|
|
1816
|
+
try:
|
|
1817
|
+
return method(
|
|
1818
|
+
dataset_name=dataset_name,
|
|
1819
|
+
dataset_id=dataset_id,
|
|
1820
|
+
columns=list(columns),
|
|
1821
|
+
rows=list(rows),
|
|
1822
|
+
metadata=dict(metadata),
|
|
1823
|
+
)
|
|
1824
|
+
except TypeError:
|
|
1825
|
+
try:
|
|
1826
|
+
return method(
|
|
1827
|
+
name=dataset_name,
|
|
1828
|
+
dataset_id=dataset_id,
|
|
1829
|
+
columns=list(columns),
|
|
1830
|
+
rows=list(rows),
|
|
1831
|
+
metadata=dict(metadata),
|
|
1832
|
+
)
|
|
1833
|
+
except TypeError:
|
|
1834
|
+
return method(
|
|
1835
|
+
dataset_name,
|
|
1836
|
+
list(columns),
|
|
1837
|
+
list(rows),
|
|
1838
|
+
)
|
|
1839
|
+
|
|
1840
|
+
if all(callable(getattr(client, name, None)) for name in ("add_columns", "add_rows")):
|
|
1841
|
+
if dataset_id is None and callable(getattr(client, "create", None)):
|
|
1842
|
+
client.create()
|
|
1843
|
+
client.add_columns(_futureagi_http_columns(columns))
|
|
1844
|
+
client.add_rows(_futureagi_http_rows(rows, columns=columns))
|
|
1845
|
+
return client
|
|
1846
|
+
|
|
1847
|
+
raise TypeError(
|
|
1848
|
+
"client must expose publish_regression_dataset(), "
|
|
1849
|
+
"create_regression_dataset(), create_dataset(), or Future AGI Dataset "
|
|
1850
|
+
"add_columns()/add_rows() methods"
|
|
1851
|
+
)
|
|
1852
|
+
|
|
1853
|
+
|
|
1854
|
+
class _FutureAGISDKDatasetPublisher:
|
|
1855
|
+
def __init__(
|
|
1856
|
+
self,
|
|
1857
|
+
*,
|
|
1858
|
+
Dataset: Any,
|
|
1859
|
+
DatasetConfig: Any,
|
|
1860
|
+
Column: Any,
|
|
1861
|
+
Row: Any,
|
|
1862
|
+
Cell: Any,
|
|
1863
|
+
DataTypeChoices: Any,
|
|
1864
|
+
ModelTypes: Any,
|
|
1865
|
+
SourceChoices: Any,
|
|
1866
|
+
fi_api_key: str,
|
|
1867
|
+
fi_secret_key: str,
|
|
1868
|
+
fi_base_url: str,
|
|
1869
|
+
timeout: float,
|
|
1870
|
+
) -> None:
|
|
1871
|
+
self.Dataset = Dataset
|
|
1872
|
+
self.DatasetConfig = DatasetConfig
|
|
1873
|
+
self.Column = Column
|
|
1874
|
+
self.Row = Row
|
|
1875
|
+
self.Cell = Cell
|
|
1876
|
+
self.DataTypeChoices = DataTypeChoices
|
|
1877
|
+
self.ModelTypes = ModelTypes
|
|
1878
|
+
self.SourceChoices = SourceChoices
|
|
1879
|
+
self.fi_api_key = fi_api_key
|
|
1880
|
+
self.fi_secret_key = fi_secret_key
|
|
1881
|
+
self.fi_base_url = fi_base_url
|
|
1882
|
+
self.timeout = timeout
|
|
1883
|
+
|
|
1884
|
+
def publish_regression_dataset(
|
|
1885
|
+
self,
|
|
1886
|
+
*,
|
|
1887
|
+
dataset_name: str,
|
|
1888
|
+
dataset_id: Optional[str],
|
|
1889
|
+
columns: Sequence[Mapping[str, Any]],
|
|
1890
|
+
rows: Sequence[Mapping[str, Any]],
|
|
1891
|
+
metadata: Mapping[str, Any],
|
|
1892
|
+
**_: Any,
|
|
1893
|
+
) -> Any:
|
|
1894
|
+
config_kwargs: dict[str, Any] = {
|
|
1895
|
+
"name": dataset_name,
|
|
1896
|
+
"model_type": self.ModelTypes.GENERATIVE_LLM,
|
|
1897
|
+
}
|
|
1898
|
+
if dataset_id:
|
|
1899
|
+
config_kwargs["id"] = dataset_id
|
|
1900
|
+
dataset_client = self.Dataset(
|
|
1901
|
+
dataset_config=self.DatasetConfig(**config_kwargs),
|
|
1902
|
+
fi_api_key=self.fi_api_key,
|
|
1903
|
+
fi_secret_key=self.fi_secret_key,
|
|
1904
|
+
fi_base_url=self.fi_base_url,
|
|
1905
|
+
timeout=self.timeout,
|
|
1906
|
+
)
|
|
1907
|
+
if dataset_id is None and _response_id(dataset_client.get_config()) is None:
|
|
1908
|
+
dataset_client = dataset_client.create()
|
|
1909
|
+
if dataset_id is None:
|
|
1910
|
+
dataset_client = dataset_client.add_columns(
|
|
1911
|
+
self._columns(columns)
|
|
1912
|
+
)
|
|
1913
|
+
if rows:
|
|
1914
|
+
dataset_client = dataset_client.add_rows(
|
|
1915
|
+
self._rows(rows, columns=columns)
|
|
1916
|
+
)
|
|
1917
|
+
config = dataset_client.get_config()
|
|
1918
|
+
return {
|
|
1919
|
+
"id": _response_id(config),
|
|
1920
|
+
"dataset_id": _response_id(config),
|
|
1921
|
+
"dataset_name": getattr(config, "name", dataset_name),
|
|
1922
|
+
"rows_added": len(rows),
|
|
1923
|
+
"columns": list(columns),
|
|
1924
|
+
"metadata": dict(metadata),
|
|
1925
|
+
}
|
|
1926
|
+
|
|
1927
|
+
def _columns(self, columns: Sequence[Mapping[str, Any]]) -> list[Any]:
|
|
1928
|
+
sdk_columns = []
|
|
1929
|
+
for column in columns:
|
|
1930
|
+
data_type = self.DataTypeChoices(str(column["data_type"]))
|
|
1931
|
+
sdk_columns.append(
|
|
1932
|
+
self.Column(
|
|
1933
|
+
name=str(column["name"]),
|
|
1934
|
+
data_type=data_type,
|
|
1935
|
+
source=self.SourceChoices.OTHERS,
|
|
1936
|
+
)
|
|
1937
|
+
)
|
|
1938
|
+
return sdk_columns
|
|
1939
|
+
|
|
1940
|
+
def _rows(
|
|
1941
|
+
self,
|
|
1942
|
+
rows: Sequence[Mapping[str, Any]],
|
|
1943
|
+
*,
|
|
1944
|
+
columns: Sequence[Mapping[str, Any]],
|
|
1945
|
+
) -> list[Any]:
|
|
1946
|
+
sdk_rows = []
|
|
1947
|
+
for index, row in enumerate(rows, start=1):
|
|
1948
|
+
cells = [
|
|
1949
|
+
self.Cell(
|
|
1950
|
+
column_name=str(column["name"]),
|
|
1951
|
+
value=_futureagi_cell_value(row.get(str(column["name"])), column),
|
|
1952
|
+
)
|
|
1953
|
+
for column in columns
|
|
1954
|
+
]
|
|
1955
|
+
sdk_rows.append(self.Row(order=index, cells=cells))
|
|
1956
|
+
return sdk_rows
|
|
1957
|
+
|
|
1958
|
+
|
|
1959
|
+
class _FutureAGIHttpDatasetPublisher:
|
|
1960
|
+
def __init__(
|
|
1961
|
+
self,
|
|
1962
|
+
*,
|
|
1963
|
+
APIKeyAuth: Any,
|
|
1964
|
+
HttpMethod: Any,
|
|
1965
|
+
RequestConfig: Any,
|
|
1966
|
+
Routes: Any,
|
|
1967
|
+
fi_api_key: str,
|
|
1968
|
+
fi_secret_key: str,
|
|
1969
|
+
fi_base_url: str,
|
|
1970
|
+
timeout: float,
|
|
1971
|
+
) -> None:
|
|
1972
|
+
self.client = APIKeyAuth(
|
|
1973
|
+
fi_api_key=fi_api_key,
|
|
1974
|
+
fi_secret_key=fi_secret_key,
|
|
1975
|
+
fi_base_url=fi_base_url,
|
|
1976
|
+
timeout=timeout,
|
|
1977
|
+
)
|
|
1978
|
+
self.HttpMethod = HttpMethod
|
|
1979
|
+
self.RequestConfig = RequestConfig
|
|
1980
|
+
self.Routes = Routes
|
|
1981
|
+
self.base_url = fi_base_url.rstrip("/")
|
|
1982
|
+
self.timeout = int(timeout)
|
|
1983
|
+
|
|
1984
|
+
def publish_regression_dataset(
|
|
1985
|
+
self,
|
|
1986
|
+
*,
|
|
1987
|
+
dataset_name: str,
|
|
1988
|
+
dataset_id: Optional[str],
|
|
1989
|
+
columns: Sequence[Mapping[str, Any]],
|
|
1990
|
+
rows: Sequence[Mapping[str, Any]],
|
|
1991
|
+
metadata: Mapping[str, Any],
|
|
1992
|
+
**_: Any,
|
|
1993
|
+
) -> dict[str, Any]:
|
|
1994
|
+
response_payload: dict[str, Any] = {}
|
|
1995
|
+
active_dataset_id = dataset_id
|
|
1996
|
+
if active_dataset_id is None:
|
|
1997
|
+
response_payload["dataset"] = self._post(
|
|
1998
|
+
str(self.Routes.dataset_empty.value),
|
|
1999
|
+
{
|
|
2000
|
+
"new_dataset_name": dataset_name,
|
|
2001
|
+
"model_type": "GenerativeLLM",
|
|
2002
|
+
"is_sdk": True,
|
|
2003
|
+
},
|
|
2004
|
+
)
|
|
2005
|
+
active_dataset_id = _response_id(response_payload["dataset"])
|
|
2006
|
+
if active_dataset_id is None:
|
|
2007
|
+
raise RuntimeError("Future AGI dataset creation did not return a dataset id")
|
|
2008
|
+
|
|
2009
|
+
if dataset_id is None:
|
|
2010
|
+
response_payload["columns"] = self._post(
|
|
2011
|
+
str(self.Routes.dataset_add_columns.value).format(
|
|
2012
|
+
dataset_id=active_dataset_id
|
|
2013
|
+
),
|
|
2014
|
+
{"new_columns_data": _futureagi_http_columns(columns)},
|
|
2015
|
+
)
|
|
2016
|
+
response_payload["rows"] = self._post(
|
|
2017
|
+
str(self.Routes.dataset_add_rows.value).format(dataset_id=active_dataset_id),
|
|
2018
|
+
{"rows": _futureagi_http_rows(rows, columns=columns)},
|
|
2019
|
+
)
|
|
2020
|
+
response_payload.update(
|
|
2021
|
+
{
|
|
2022
|
+
"id": active_dataset_id,
|
|
2023
|
+
"dataset_id": active_dataset_id,
|
|
2024
|
+
"dataset_name": dataset_name,
|
|
2025
|
+
"rows_added": len(rows),
|
|
2026
|
+
"metadata": dict(metadata),
|
|
2027
|
+
}
|
|
2028
|
+
)
|
|
2029
|
+
return response_payload
|
|
2030
|
+
|
|
2031
|
+
def _post(self, route: str, payload: Mapping[str, Any]) -> dict[str, Any]:
|
|
2032
|
+
response = self.client.request(
|
|
2033
|
+
config=self.RequestConfig(
|
|
2034
|
+
method=self.HttpMethod.POST,
|
|
2035
|
+
url=f"{self.base_url}/{route}",
|
|
2036
|
+
json=dict(payload),
|
|
2037
|
+
timeout=self.timeout,
|
|
2038
|
+
)
|
|
2039
|
+
)
|
|
2040
|
+
return _http_response_payload(response)
|
|
2041
|
+
|
|
2042
|
+
|
|
2043
|
+
class _FutureAGIHttpDatasetReader:
|
|
2044
|
+
def __init__(
|
|
2045
|
+
self,
|
|
2046
|
+
*,
|
|
2047
|
+
APIKeyAuth: Any,
|
|
2048
|
+
HttpMethod: Any,
|
|
2049
|
+
RequestConfig: Any,
|
|
2050
|
+
Routes: Any,
|
|
2051
|
+
fi_api_key: str,
|
|
2052
|
+
fi_secret_key: str,
|
|
2053
|
+
fi_base_url: str,
|
|
2054
|
+
timeout: float,
|
|
2055
|
+
) -> None:
|
|
2056
|
+
self.client = APIKeyAuth(
|
|
2057
|
+
fi_api_key=fi_api_key,
|
|
2058
|
+
fi_secret_key=fi_secret_key,
|
|
2059
|
+
fi_base_url=fi_base_url,
|
|
2060
|
+
timeout=timeout,
|
|
2061
|
+
)
|
|
2062
|
+
self.HttpMethod = HttpMethod
|
|
2063
|
+
self.RequestConfig = RequestConfig
|
|
2064
|
+
self.Routes = Routes
|
|
2065
|
+
self.base_url = fi_base_url.rstrip("/")
|
|
2066
|
+
self.timeout = int(timeout)
|
|
2067
|
+
|
|
2068
|
+
def fetch_regression_dataset(
|
|
2069
|
+
self,
|
|
2070
|
+
*,
|
|
2071
|
+
dataset_id: str,
|
|
2072
|
+
page_size: int,
|
|
2073
|
+
current_page_index: int,
|
|
2074
|
+
) -> dict[str, Any]:
|
|
2075
|
+
route = str(self.Routes.dataset_table.value).format(dataset_id=dataset_id)
|
|
2076
|
+
response = self.client.request(
|
|
2077
|
+
config=self.RequestConfig(
|
|
2078
|
+
method=self.HttpMethod.GET,
|
|
2079
|
+
url=f"{self.base_url}/{route}",
|
|
2080
|
+
params={
|
|
2081
|
+
"page_size": page_size,
|
|
2082
|
+
"current_page_index": current_page_index,
|
|
2083
|
+
},
|
|
2084
|
+
timeout=self.timeout,
|
|
2085
|
+
)
|
|
2086
|
+
)
|
|
2087
|
+
return _http_response_payload(response)
|
|
2088
|
+
|
|
2089
|
+
|
|
2090
|
+
class _FutureAGIHttpExperimentReader:
|
|
2091
|
+
def __init__(
|
|
2092
|
+
self,
|
|
2093
|
+
*,
|
|
2094
|
+
APIKeyAuth: Any,
|
|
2095
|
+
HttpMethod: Any,
|
|
2096
|
+
RequestConfig: Any,
|
|
2097
|
+
fi_api_key: str,
|
|
2098
|
+
fi_secret_key: str,
|
|
2099
|
+
fi_base_url: str,
|
|
2100
|
+
timeout: float,
|
|
2101
|
+
) -> None:
|
|
2102
|
+
self.client = APIKeyAuth(
|
|
2103
|
+
fi_api_key=fi_api_key,
|
|
2104
|
+
fi_secret_key=fi_secret_key,
|
|
2105
|
+
fi_base_url=fi_base_url,
|
|
2106
|
+
timeout=timeout,
|
|
2107
|
+
)
|
|
2108
|
+
self.HttpMethod = HttpMethod
|
|
2109
|
+
self.RequestConfig = RequestConfig
|
|
2110
|
+
self.base_url = fi_base_url.rstrip("/")
|
|
2111
|
+
self.timeout = int(timeout)
|
|
2112
|
+
|
|
2113
|
+
def fetch_experiment_history(
|
|
2114
|
+
self,
|
|
2115
|
+
*,
|
|
2116
|
+
experiment_id: str,
|
|
2117
|
+
page_size: int,
|
|
2118
|
+
max_pages: int,
|
|
2119
|
+
include_rows: bool,
|
|
2120
|
+
include_stats: bool,
|
|
2121
|
+
prefer_v2: bool,
|
|
2122
|
+
) -> dict[str, Any]:
|
|
2123
|
+
history: dict[str, Any] = {"experiment_id": experiment_id}
|
|
2124
|
+
try:
|
|
2125
|
+
history["detail"] = self._get_first(
|
|
2126
|
+
[
|
|
2127
|
+
f"model-hub/experiments/v2/{experiment_id}/",
|
|
2128
|
+
"model-hub/experiments/",
|
|
2129
|
+
]
|
|
2130
|
+
if prefer_v2
|
|
2131
|
+
else [
|
|
2132
|
+
"model-hub/experiments/",
|
|
2133
|
+
f"model-hub/experiments/v2/{experiment_id}/",
|
|
2134
|
+
],
|
|
2135
|
+
params={"experiment_id": experiment_id},
|
|
2136
|
+
)
|
|
2137
|
+
except Exception as exc:
|
|
2138
|
+
history["detail_error"] = str(exc)
|
|
2139
|
+
if include_stats:
|
|
2140
|
+
try:
|
|
2141
|
+
history["stats"] = self._get_first(
|
|
2142
|
+
[
|
|
2143
|
+
f"model-hub/experiments/v2/{experiment_id}/stats/",
|
|
2144
|
+
f"model-hub/experiments/{experiment_id}/stats/",
|
|
2145
|
+
]
|
|
2146
|
+
if prefer_v2
|
|
2147
|
+
else [
|
|
2148
|
+
f"model-hub/experiments/{experiment_id}/stats/",
|
|
2149
|
+
f"model-hub/experiments/v2/{experiment_id}/stats/",
|
|
2150
|
+
]
|
|
2151
|
+
)
|
|
2152
|
+
except Exception as exc:
|
|
2153
|
+
history["stats_error"] = str(exc)
|
|
2154
|
+
if include_rows:
|
|
2155
|
+
row_routes = (
|
|
2156
|
+
[
|
|
2157
|
+
f"model-hub/experiments/v2/{experiment_id}/rows/",
|
|
2158
|
+
f"model-hub/experiments/{experiment_id}/",
|
|
2159
|
+
]
|
|
2160
|
+
if prefer_v2
|
|
2161
|
+
else [
|
|
2162
|
+
f"model-hub/experiments/{experiment_id}/",
|
|
2163
|
+
f"model-hub/experiments/v2/{experiment_id}/rows/",
|
|
2164
|
+
]
|
|
2165
|
+
)
|
|
2166
|
+
pages = []
|
|
2167
|
+
try:
|
|
2168
|
+
for page_index in range(max_pages):
|
|
2169
|
+
payload = self._get_first(
|
|
2170
|
+
row_routes,
|
|
2171
|
+
params={
|
|
2172
|
+
"page_size": page_size,
|
|
2173
|
+
"current_page_index": page_index,
|
|
2174
|
+
},
|
|
2175
|
+
)
|
|
2176
|
+
pages.append(payload)
|
|
2177
|
+
result = _futureagi_payload_result(payload)
|
|
2178
|
+
total_pages = _futureagi_total_pages(result)
|
|
2179
|
+
row_count = len(_futureagi_table_rows(result))
|
|
2180
|
+
if total_pages is not None:
|
|
2181
|
+
if page_index + 1 >= total_pages:
|
|
2182
|
+
break
|
|
2183
|
+
elif row_count < page_size:
|
|
2184
|
+
break
|
|
2185
|
+
else:
|
|
2186
|
+
break
|
|
2187
|
+
history["rows"] = pages
|
|
2188
|
+
except Exception as exc:
|
|
2189
|
+
history["rows_error"] = str(exc)
|
|
2190
|
+
if "stats" not in history and "rows" not in history:
|
|
2191
|
+
try:
|
|
2192
|
+
history["list"] = self._get(
|
|
2193
|
+
"model-hub/experiments/data/",
|
|
2194
|
+
params={"page_size": page_size, "current_page_index": 0},
|
|
2195
|
+
)
|
|
2196
|
+
except Exception as exc:
|
|
2197
|
+
history["list_error"] = str(exc)
|
|
2198
|
+
return history
|
|
2199
|
+
|
|
2200
|
+
def _get_first(
|
|
2201
|
+
self,
|
|
2202
|
+
routes: Sequence[str],
|
|
2203
|
+
params: Optional[Mapping[str, Any]] = None,
|
|
2204
|
+
) -> dict[str, Any]:
|
|
2205
|
+
errors: list[str] = []
|
|
2206
|
+
for route in routes:
|
|
2207
|
+
try:
|
|
2208
|
+
return self._get(route, params=params)
|
|
2209
|
+
except Exception as exc:
|
|
2210
|
+
errors.append(f"{route}: {exc}")
|
|
2211
|
+
raise RuntimeError("; ".join(errors))
|
|
2212
|
+
|
|
2213
|
+
def _get(
|
|
2214
|
+
self,
|
|
2215
|
+
route: str,
|
|
2216
|
+
params: Optional[Mapping[str, Any]] = None,
|
|
2217
|
+
) -> dict[str, Any]:
|
|
2218
|
+
response = self.client.request(
|
|
2219
|
+
config=self.RequestConfig(
|
|
2220
|
+
method=self.HttpMethod.GET,
|
|
2221
|
+
url=f"{self.base_url}/{route}",
|
|
2222
|
+
params=dict(params or {}),
|
|
2223
|
+
timeout=self.timeout,
|
|
2224
|
+
)
|
|
2225
|
+
)
|
|
2226
|
+
return _http_response_payload(response)
|
|
2227
|
+
|
|
2228
|
+
|
|
2229
|
+
def _futureagi_http_columns(
|
|
2230
|
+
columns: Sequence[Mapping[str, Any]],
|
|
2231
|
+
) -> list[dict[str, Any]]:
|
|
2232
|
+
return [
|
|
2233
|
+
{
|
|
2234
|
+
"name": str(column["name"]),
|
|
2235
|
+
"data_type": str(column["data_type"]),
|
|
2236
|
+
"source": "OTHERS",
|
|
2237
|
+
}
|
|
2238
|
+
for column in columns
|
|
2239
|
+
]
|
|
2240
|
+
|
|
2241
|
+
|
|
2242
|
+
def _futureagi_http_rows(
|
|
2243
|
+
rows: Sequence[Mapping[str, Any]],
|
|
2244
|
+
*,
|
|
2245
|
+
columns: Sequence[Mapping[str, Any]],
|
|
2246
|
+
) -> list[dict[str, Any]]:
|
|
2247
|
+
return [
|
|
2248
|
+
{
|
|
2249
|
+
"order": index,
|
|
2250
|
+
"cells": [
|
|
2251
|
+
{
|
|
2252
|
+
"column_name": str(column["name"]),
|
|
2253
|
+
"value": _futureagi_cell_value(
|
|
2254
|
+
row.get(str(column["name"])),
|
|
2255
|
+
column,
|
|
2256
|
+
),
|
|
2257
|
+
}
|
|
2258
|
+
for column in columns
|
|
2259
|
+
],
|
|
2260
|
+
}
|
|
2261
|
+
for index, row in enumerate(rows, start=1)
|
|
2262
|
+
]
|
|
2263
|
+
|
|
2264
|
+
|
|
2265
|
+
def _futureagi_cell_value(value: Any, column: Mapping[str, Any]) -> Any:
|
|
2266
|
+
if value is None:
|
|
2267
|
+
return ""
|
|
2268
|
+
data_type = str(column.get("data_type") or "text")
|
|
2269
|
+
if data_type in {"json", "array"}:
|
|
2270
|
+
return json.dumps(value, sort_keys=True, default=str)
|
|
2271
|
+
if isinstance(value, str):
|
|
2272
|
+
return value
|
|
2273
|
+
return json.dumps(value, sort_keys=True, default=str)
|
|
2274
|
+
|
|
2275
|
+
|
|
2276
|
+
def _http_response_payload(response: Any) -> dict[str, Any]:
|
|
2277
|
+
status_code = getattr(response, "status_code", None)
|
|
2278
|
+
ok = getattr(response, "ok", True)
|
|
2279
|
+
text = getattr(response, "text", "")
|
|
2280
|
+
if ok is False:
|
|
2281
|
+
raise RuntimeError(f"Future AGI API request failed with status {status_code}: {text[:500]}")
|
|
2282
|
+
try:
|
|
2283
|
+
payload = response.json()
|
|
2284
|
+
except Exception:
|
|
2285
|
+
return {"status_code": status_code, "text": text}
|
|
2286
|
+
return dict(payload) if isinstance(payload, Mapping) else {"value": payload}
|
|
2287
|
+
|
|
2288
|
+
|
|
2289
|
+
def _dataset_sink_result(
|
|
2290
|
+
*,
|
|
2291
|
+
provider: str,
|
|
2292
|
+
dataset_name: str,
|
|
2293
|
+
case_count: int,
|
|
2294
|
+
status: str,
|
|
2295
|
+
dataset_id: Optional[str] = None,
|
|
2296
|
+
endpoint: Optional[str] = None,
|
|
2297
|
+
dry_run: bool = False,
|
|
2298
|
+
failures: Optional[Sequence[str]] = None,
|
|
2299
|
+
response: Optional[Mapping[str, Any]] = None,
|
|
2300
|
+
metadata: Optional[Mapping[str, Any]] = None,
|
|
2301
|
+
) -> AgentDatasetSinkResult:
|
|
2302
|
+
return AgentDatasetSinkResult(
|
|
2303
|
+
provider=provider,
|
|
2304
|
+
dataset_name=dataset_name,
|
|
2305
|
+
dataset_id=dataset_id,
|
|
2306
|
+
case_count=case_count,
|
|
2307
|
+
endpoint=endpoint,
|
|
2308
|
+
dry_run=dry_run,
|
|
2309
|
+
status=status,
|
|
2310
|
+
failures=list(failures or []),
|
|
2311
|
+
response=dict(response or {}),
|
|
2312
|
+
metadata=dict(metadata or {}),
|
|
2313
|
+
)
|
|
2314
|
+
|
|
2315
|
+
|
|
2316
|
+
def _response_id(value: Any) -> Optional[str]:
|
|
2317
|
+
if value is None:
|
|
2318
|
+
return None
|
|
2319
|
+
if isinstance(value, Mapping):
|
|
2320
|
+
for key in ("id", "dataset_id", "datasetId"):
|
|
2321
|
+
if value.get(key) is not None:
|
|
2322
|
+
return str(value[key])
|
|
2323
|
+
for key in ("result", "data", "dataset"):
|
|
2324
|
+
nested = value.get(key)
|
|
2325
|
+
nested_id = _response_id(nested)
|
|
2326
|
+
if nested_id is not None:
|
|
2327
|
+
return nested_id
|
|
2328
|
+
return None
|
|
2329
|
+
for key in ("id", "dataset_id", "datasetId"):
|
|
2330
|
+
item = getattr(value, key, None)
|
|
2331
|
+
if item is not None:
|
|
2332
|
+
return str(item)
|
|
2333
|
+
return None
|
|
2334
|
+
|
|
2335
|
+
|
|
2336
|
+
def _safe_response_payload(value: Any) -> dict[str, Any]:
|
|
2337
|
+
if value is None:
|
|
2338
|
+
return {}
|
|
2339
|
+
if hasattr(value, "model_dump"):
|
|
2340
|
+
value = value.model_dump()
|
|
2341
|
+
elif hasattr(value, "dict"):
|
|
2342
|
+
value = value.dict()
|
|
2343
|
+
if isinstance(value, Mapping):
|
|
2344
|
+
return dict(value)
|
|
2345
|
+
if isinstance(value, Sequence) and not isinstance(value, (str, bytes, bytearray)):
|
|
2346
|
+
return {"items": list(value)}
|
|
2347
|
+
payload: dict[str, Any] = {}
|
|
2348
|
+
response_id = _response_id(value)
|
|
2349
|
+
if response_id is not None:
|
|
2350
|
+
payload["id"] = response_id
|
|
2351
|
+
if not payload:
|
|
2352
|
+
payload["repr"] = str(value)
|
|
2353
|
+
return payload
|
|
2354
|
+
|
|
2355
|
+
|
|
2356
|
+
def _optional_str(value: Any) -> Optional[str]:
|
|
2357
|
+
if value is None:
|
|
2358
|
+
return None
|
|
2359
|
+
text = str(value).strip()
|
|
2360
|
+
return text or None
|
|
2361
|
+
|
|
2362
|
+
|
|
2363
|
+
def _first_float(*values: Any) -> float:
|
|
2364
|
+
for value in values:
|
|
2365
|
+
if value is None:
|
|
2366
|
+
continue
|
|
2367
|
+
try:
|
|
2368
|
+
return float(value)
|
|
2369
|
+
except (TypeError, ValueError):
|
|
2370
|
+
continue
|
|
2371
|
+
return 0.0
|
|
2372
|
+
|
|
2373
|
+
|
|
2374
|
+
def _registry_replay_case_signature(case_ids: Sequence[str]) -> str:
|
|
2375
|
+
payload = "\n".join(sorted(str(case_id) for case_id in case_ids))
|
|
2376
|
+
return hashlib.sha256(payload.encode("utf-8")).hexdigest()
|
|
2377
|
+
|
|
2378
|
+
|
|
2379
|
+
def _registry_replay_retention_key(
|
|
2380
|
+
*,
|
|
2381
|
+
registry_version: str,
|
|
2382
|
+
dataset_name: str,
|
|
2383
|
+
dataset_id: Optional[str],
|
|
2384
|
+
case_signature: str,
|
|
2385
|
+
) -> str:
|
|
2386
|
+
dataset_key = dataset_id or dataset_name
|
|
2387
|
+
return f"{registry_version}:{dataset_key}:{case_signature[:16]}"
|
|
2388
|
+
|
|
2389
|
+
|
|
2390
|
+
def _registry_replay_requirements(
|
|
2391
|
+
selection: Mapping[str, Any],
|
|
2392
|
+
) -> tuple[list[str], list[str]]:
|
|
2393
|
+
presets: set[str] = set()
|
|
2394
|
+
families: set[str] = set()
|
|
2395
|
+
for item in _sequence_items(selection.get("required")):
|
|
2396
|
+
if not isinstance(item, Mapping):
|
|
2397
|
+
continue
|
|
2398
|
+
preset = _optional_str(item.get("preset"))
|
|
2399
|
+
family = _optional_str(item.get("invariant_family"))
|
|
2400
|
+
if preset:
|
|
2401
|
+
presets.add(preset)
|
|
2402
|
+
if family:
|
|
2403
|
+
families.add(family)
|
|
2404
|
+
for item in _sequence_items(selection.get("selected")):
|
|
2405
|
+
if not isinstance(item, Mapping):
|
|
2406
|
+
continue
|
|
2407
|
+
preset = _optional_str(item.get("preset"))
|
|
2408
|
+
family = _optional_str(item.get("invariant_family"))
|
|
2409
|
+
if preset:
|
|
2410
|
+
presets.add(preset)
|
|
2411
|
+
if family:
|
|
2412
|
+
families.add(family)
|
|
2413
|
+
return sorted(presets), sorted(families)
|
|
2414
|
+
|
|
2415
|
+
|
|
2416
|
+
def _coerce_registry_replay_manifest(
|
|
2417
|
+
value: AgentRegistryReplayPackManifest | Mapping[str, Any],
|
|
2418
|
+
) -> AgentRegistryReplayPackManifest:
|
|
2419
|
+
if isinstance(value, AgentRegistryReplayPackManifest):
|
|
2420
|
+
return value
|
|
2421
|
+
if isinstance(value, Mapping):
|
|
2422
|
+
return AgentRegistryReplayPackManifest(**dict(value))
|
|
2423
|
+
raise TypeError("manifest must be AgentRegistryReplayPackManifest or mapping.")
|
|
2424
|
+
|
|
2425
|
+
|
|
2426
|
+
def _coerce_registry_replay_promotion_check(
|
|
2427
|
+
value: AgentRegistryReplayPackPromotionCheck | Mapping[str, Any],
|
|
2428
|
+
) -> AgentRegistryReplayPackPromotionCheck:
|
|
2429
|
+
if isinstance(value, AgentRegistryReplayPackPromotionCheck):
|
|
2430
|
+
return value
|
|
2431
|
+
if isinstance(value, Mapping):
|
|
2432
|
+
return AgentRegistryReplayPackPromotionCheck(**dict(value))
|
|
2433
|
+
raise TypeError(
|
|
2434
|
+
"promotion check must be AgentRegistryReplayPackPromotionCheck or mapping."
|
|
2435
|
+
)
|
|
2436
|
+
|
|
2437
|
+
|
|
2438
|
+
def _coerce_registry_replay_lineage_report(
|
|
2439
|
+
value: AgentRegistryReplayPackLineageReport | Mapping[str, Any],
|
|
2440
|
+
) -> AgentRegistryReplayPackLineageReport:
|
|
2441
|
+
if isinstance(value, AgentRegistryReplayPackLineageReport):
|
|
2442
|
+
return value
|
|
2443
|
+
if isinstance(value, Mapping):
|
|
2444
|
+
return AgentRegistryReplayPackLineageReport(**dict(value))
|
|
2445
|
+
raise TypeError(
|
|
2446
|
+
"lineage must be AgentRegistryReplayPackLineageReport or mapping."
|
|
2447
|
+
)
|
|
2448
|
+
|
|
2449
|
+
|
|
2450
|
+
def _registry_replay_promotion_checks_by_key(
|
|
2451
|
+
promotion_checks: Optional[
|
|
2452
|
+
Sequence[AgentRegistryReplayPackPromotionCheck | Mapping[str, Any]]
|
|
2453
|
+
| Mapping[str, AgentRegistryReplayPackPromotionCheck | Mapping[str, Any]]
|
|
2454
|
+
],
|
|
2455
|
+
) -> dict[str, AgentRegistryReplayPackPromotionCheck]:
|
|
2456
|
+
checks_by_key: dict[str, AgentRegistryReplayPackPromotionCheck] = {}
|
|
2457
|
+
if promotion_checks is None:
|
|
2458
|
+
return checks_by_key
|
|
2459
|
+
if isinstance(promotion_checks, Mapping):
|
|
2460
|
+
iterable = promotion_checks.values()
|
|
2461
|
+
explicit_keys = [str(key) for key in promotion_checks.keys()]
|
|
2462
|
+
else:
|
|
2463
|
+
iterable = promotion_checks
|
|
2464
|
+
explicit_keys = []
|
|
2465
|
+
for index, raw_check in enumerate(iterable):
|
|
2466
|
+
check = _coerce_registry_replay_promotion_check(raw_check)
|
|
2467
|
+
keys = [
|
|
2468
|
+
check.manifest.retention_key,
|
|
2469
|
+
check.dataset_id,
|
|
2470
|
+
check.registry_version,
|
|
2471
|
+
check.manifest.dataset_id,
|
|
2472
|
+
check.manifest.dataset_name,
|
|
2473
|
+
]
|
|
2474
|
+
if index < len(explicit_keys):
|
|
2475
|
+
keys.append(explicit_keys[index])
|
|
2476
|
+
for key in keys:
|
|
2477
|
+
if key:
|
|
2478
|
+
checks_by_key[str(key)] = check
|
|
2479
|
+
return checks_by_key
|
|
2480
|
+
|
|
2481
|
+
|
|
2482
|
+
def _registry_replay_lineage_entry(
|
|
2483
|
+
manifest: AgentRegistryReplayPackManifest,
|
|
2484
|
+
*,
|
|
2485
|
+
checks_by_key: Mapping[str, AgentRegistryReplayPackPromotionCheck],
|
|
2486
|
+
) -> AgentRegistryReplayPackLineageEntry:
|
|
2487
|
+
check = _registry_replay_check_for_manifest(manifest, checks_by_key)
|
|
2488
|
+
check_metadata = _ensure_mapping(check.metadata if check else {})
|
|
2489
|
+
optimizer_backend = _optional_str(
|
|
2490
|
+
check_metadata.get("optimizer_backend")
|
|
2491
|
+
or check_metadata.get("selected_optimizer")
|
|
2492
|
+
or check_metadata.get("optimizer")
|
|
2493
|
+
)
|
|
2494
|
+
selected_patch_signature = _optional_str(
|
|
2495
|
+
check_metadata.get("selected_patch_signature")
|
|
2496
|
+
or check_metadata.get("selected_candidate_signature")
|
|
2497
|
+
or check_metadata.get("best_candidate_signature")
|
|
2498
|
+
)
|
|
2499
|
+
readback_signature_matches = None
|
|
2500
|
+
if check is not None:
|
|
2501
|
+
readback_signature_matches = (
|
|
2502
|
+
check.loaded_case_signature == check.expected_case_signature
|
|
2503
|
+
)
|
|
2504
|
+
return AgentRegistryReplayPackLineageEntry(
|
|
2505
|
+
registry_version=manifest.registry_version,
|
|
2506
|
+
dataset_name=manifest.dataset_name,
|
|
2507
|
+
dataset_id=manifest.dataset_id,
|
|
2508
|
+
retention_key=manifest.retention_key,
|
|
2509
|
+
case_count=manifest.case_count,
|
|
2510
|
+
case_signature=manifest.case_signature,
|
|
2511
|
+
coverage_score=manifest.coverage_score,
|
|
2512
|
+
selection_complete=manifest.selection_complete,
|
|
2513
|
+
required_presets=list(manifest.required_presets),
|
|
2514
|
+
required_invariant_families=list(manifest.required_invariant_families),
|
|
2515
|
+
promotion_promotable=check.promotable if check else None,
|
|
2516
|
+
loaded_case_count=check.loaded_case_count if check else None,
|
|
2517
|
+
replay_record_count=check.replay_record_count if check else None,
|
|
2518
|
+
readback_signature_matches=readback_signature_matches,
|
|
2519
|
+
optimizer_score=check.optimizer_score if check else None,
|
|
2520
|
+
min_optimizer_score=check.min_optimizer_score if check else None,
|
|
2521
|
+
optimizer_backend=optimizer_backend,
|
|
2522
|
+
selected_patch_signature=selected_patch_signature,
|
|
2523
|
+
failures=list(check.failures) if check else [],
|
|
2524
|
+
metadata={
|
|
2525
|
+
"manifest_name": manifest.name,
|
|
2526
|
+
"selected_positive_count": manifest.selected_positive_count,
|
|
2527
|
+
"selected_negative_count": manifest.selected_negative_count,
|
|
2528
|
+
"check_metadata": copy.deepcopy(check_metadata),
|
|
2529
|
+
},
|
|
2530
|
+
)
|
|
2531
|
+
|
|
2532
|
+
|
|
2533
|
+
def _registry_replay_check_for_manifest(
|
|
2534
|
+
manifest: AgentRegistryReplayPackManifest,
|
|
2535
|
+
checks_by_key: Mapping[str, AgentRegistryReplayPackPromotionCheck],
|
|
2536
|
+
) -> Optional[AgentRegistryReplayPackPromotionCheck]:
|
|
2537
|
+
for key in (
|
|
2538
|
+
manifest.retention_key,
|
|
2539
|
+
manifest.dataset_id,
|
|
2540
|
+
manifest.registry_version,
|
|
2541
|
+
manifest.dataset_name,
|
|
2542
|
+
):
|
|
2543
|
+
if key and str(key) in checks_by_key:
|
|
2544
|
+
return checks_by_key[str(key)]
|
|
2545
|
+
return None
|
|
2546
|
+
|
|
2547
|
+
|
|
2548
|
+
def _registry_replay_lineage_transition(
|
|
2549
|
+
previous: AgentRegistryReplayPackLineageEntry,
|
|
2550
|
+
current: AgentRegistryReplayPackLineageEntry,
|
|
2551
|
+
) -> AgentRegistryReplayPackLineageTransition:
|
|
2552
|
+
coverage_delta = round(current.coverage_score - previous.coverage_score, 8)
|
|
2553
|
+
optimizer_score_delta: Optional[float] = None
|
|
2554
|
+
if previous.optimizer_score is not None and current.optimizer_score is not None:
|
|
2555
|
+
optimizer_score_delta = round(
|
|
2556
|
+
current.optimizer_score - previous.optimizer_score,
|
|
2557
|
+
8,
|
|
2558
|
+
)
|
|
2559
|
+
selected_patch_changed = _optional_change(
|
|
2560
|
+
previous.selected_patch_signature,
|
|
2561
|
+
current.selected_patch_signature,
|
|
2562
|
+
)
|
|
2563
|
+
optimizer_backend_changed = _optional_change(
|
|
2564
|
+
previous.optimizer_backend,
|
|
2565
|
+
current.optimizer_backend,
|
|
2566
|
+
)
|
|
2567
|
+
promotion_status_changed = _optional_bool_change(
|
|
2568
|
+
previous.promotion_promotable,
|
|
2569
|
+
current.promotion_promotable,
|
|
2570
|
+
)
|
|
2571
|
+
previous_presets = set(previous.required_presets)
|
|
2572
|
+
current_presets = set(current.required_presets)
|
|
2573
|
+
previous_families = set(previous.required_invariant_families)
|
|
2574
|
+
current_families = set(current.required_invariant_families)
|
|
2575
|
+
drift_reasons: list[str] = []
|
|
2576
|
+
if previous.case_signature != current.case_signature:
|
|
2577
|
+
drift_reasons.append("case_signature_changed")
|
|
2578
|
+
if previous.retention_key != current.retention_key:
|
|
2579
|
+
drift_reasons.append("retention_key_changed")
|
|
2580
|
+
if coverage_delta != 0:
|
|
2581
|
+
drift_reasons.append("coverage_score_changed")
|
|
2582
|
+
if optimizer_score_delta is not None and optimizer_score_delta != 0:
|
|
2583
|
+
drift_reasons.append("optimizer_score_changed")
|
|
2584
|
+
if selected_patch_changed:
|
|
2585
|
+
drift_reasons.append("selected_patch_changed")
|
|
2586
|
+
if optimizer_backend_changed:
|
|
2587
|
+
drift_reasons.append("optimizer_backend_changed")
|
|
2588
|
+
if promotion_status_changed:
|
|
2589
|
+
drift_reasons.append("promotion_status_changed")
|
|
2590
|
+
if previous_presets != current_presets:
|
|
2591
|
+
drift_reasons.append("required_presets_changed")
|
|
2592
|
+
if previous_families != current_families:
|
|
2593
|
+
drift_reasons.append("required_invariant_families_changed")
|
|
2594
|
+
return AgentRegistryReplayPackLineageTransition(
|
|
2595
|
+
from_registry_version=previous.registry_version,
|
|
2596
|
+
to_registry_version=current.registry_version,
|
|
2597
|
+
from_dataset_id=previous.dataset_id,
|
|
2598
|
+
to_dataset_id=current.dataset_id,
|
|
2599
|
+
case_count_delta=current.case_count - previous.case_count,
|
|
2600
|
+
coverage_delta=coverage_delta,
|
|
2601
|
+
optimizer_score_delta=optimizer_score_delta,
|
|
2602
|
+
case_signature_changed=previous.case_signature != current.case_signature,
|
|
2603
|
+
retention_key_changed=previous.retention_key != current.retention_key,
|
|
2604
|
+
selected_patch_changed=selected_patch_changed,
|
|
2605
|
+
optimizer_backend_changed=optimizer_backend_changed,
|
|
2606
|
+
promotion_status_changed=promotion_status_changed,
|
|
2607
|
+
added_required_presets=sorted(current_presets - previous_presets),
|
|
2608
|
+
removed_required_presets=sorted(previous_presets - current_presets),
|
|
2609
|
+
added_invariant_families=sorted(current_families - previous_families),
|
|
2610
|
+
removed_invariant_families=sorted(previous_families - current_families),
|
|
2611
|
+
drift_reasons=drift_reasons,
|
|
2612
|
+
)
|
|
2613
|
+
|
|
2614
|
+
|
|
2615
|
+
def _optional_change(previous: Optional[str], current: Optional[str]) -> Optional[bool]:
|
|
2616
|
+
if previous is None or current is None:
|
|
2617
|
+
return None
|
|
2618
|
+
return previous != current
|
|
2619
|
+
|
|
2620
|
+
|
|
2621
|
+
def _optional_bool_change(
|
|
2622
|
+
previous: Optional[bool],
|
|
2623
|
+
current: Optional[bool],
|
|
2624
|
+
) -> Optional[bool]:
|
|
2625
|
+
if previous is None or current is None:
|
|
2626
|
+
return None
|
|
2627
|
+
return previous != current
|
|
2628
|
+
|
|
2629
|
+
|
|
2630
|
+
def _unique_strings(values: Sequence[str]) -> list[str]:
|
|
2631
|
+
return list(dict.fromkeys(str(value) for value in values if value))
|
|
2632
|
+
|
|
2633
|
+
|
|
2634
|
+
def _registry_replay_triage_severity(
|
|
2635
|
+
*,
|
|
2636
|
+
blocking_reasons: Sequence[str],
|
|
2637
|
+
warnings: Sequence[str],
|
|
2638
|
+
) -> str:
|
|
2639
|
+
critical = {
|
|
2640
|
+
"futureagi_readback_signature_mismatch",
|
|
2641
|
+
"futureagi_readback_case_count_mismatch",
|
|
2642
|
+
"futureagi_replay_record_count_mismatch",
|
|
2643
|
+
"latest_promotion_failed",
|
|
2644
|
+
"latest_promotion_failures",
|
|
2645
|
+
"missing_latest_promotion_check",
|
|
2646
|
+
"missing_latest_optimizer_score",
|
|
2647
|
+
}
|
|
2648
|
+
high = {
|
|
2649
|
+
"coverage_regression",
|
|
2650
|
+
"optimizer_score_regression",
|
|
2651
|
+
"latest_below_best_optimizer_score",
|
|
2652
|
+
"required_presets_removed",
|
|
2653
|
+
"required_invariant_families_removed",
|
|
2654
|
+
}
|
|
2655
|
+
if blocking_reasons:
|
|
2656
|
+
if any(reason in critical for reason in blocking_reasons):
|
|
2657
|
+
return "critical"
|
|
2658
|
+
if any(reason in high for reason in blocking_reasons):
|
|
2659
|
+
return "high"
|
|
2660
|
+
return "medium"
|
|
2661
|
+
if warnings:
|
|
2662
|
+
medium = {
|
|
2663
|
+
"case_signature_changed",
|
|
2664
|
+
"selected_patch_changed",
|
|
2665
|
+
"optimizer_backend_changed",
|
|
2666
|
+
"required_contract_expanded",
|
|
2667
|
+
"optimizer_score_drift",
|
|
2668
|
+
}
|
|
2669
|
+
if any(warning in medium for warning in warnings):
|
|
2670
|
+
return "medium"
|
|
2671
|
+
return "low"
|
|
2672
|
+
return "none"
|
|
2673
|
+
|
|
2674
|
+
|
|
2675
|
+
def _registry_replay_triage_recommendations(
|
|
2676
|
+
*,
|
|
2677
|
+
blocking_reasons: Sequence[str],
|
|
2678
|
+
warnings: Sequence[str],
|
|
2679
|
+
) -> list[str]:
|
|
2680
|
+
recommendation_by_reason = {
|
|
2681
|
+
"missing_latest_promotion_check": (
|
|
2682
|
+
"Run check_futureagi_registry_replay_pack_promotion() with Future AGI "
|
|
2683
|
+
"readback and optimizer replay before rollout."
|
|
2684
|
+
),
|
|
2685
|
+
"missing_latest_optimizer_score": (
|
|
2686
|
+
"Attach optimizer replay score evidence before rollout."
|
|
2687
|
+
),
|
|
2688
|
+
"latest_promotion_failed": (
|
|
2689
|
+
"Fix the latest promotion-gate failures before rollout."
|
|
2690
|
+
),
|
|
2691
|
+
"latest_promotion_failures": (
|
|
2692
|
+
"Inspect latest promotion failures and repair the replay pack or candidate."
|
|
2693
|
+
),
|
|
2694
|
+
"futureagi_readback_signature_mismatch": (
|
|
2695
|
+
"Republish or reload the Future AGI dataset until readback case signature "
|
|
2696
|
+
"matches the manifest."
|
|
2697
|
+
),
|
|
2698
|
+
"futureagi_readback_case_count_mismatch": (
|
|
2699
|
+
"Republish or reload the Future AGI dataset until readback case count "
|
|
2700
|
+
"matches the manifest."
|
|
2701
|
+
),
|
|
2702
|
+
"futureagi_replay_record_count_mismatch": (
|
|
2703
|
+
"Rebuild replay rows so every Future AGI dataset row becomes one "
|
|
2704
|
+
"optimizer replay record."
|
|
2705
|
+
),
|
|
2706
|
+
"coverage_regression": (
|
|
2707
|
+
"Block rollout until replay-pack coverage recovers or the registry "
|
|
2708
|
+
"coverage loss is explicitly approved."
|
|
2709
|
+
),
|
|
2710
|
+
"optimizer_score_regression": (
|
|
2711
|
+
"Block rollout and rerun curriculum or multi_interaction optimization "
|
|
2712
|
+
"against the latest Future AGI replay pack."
|
|
2713
|
+
),
|
|
2714
|
+
"latest_below_best_optimizer_score": (
|
|
2715
|
+
"Compare the latest candidate with the best historical replay candidate "
|
|
2716
|
+
"before promotion."
|
|
2717
|
+
),
|
|
2718
|
+
"required_presets_removed": (
|
|
2719
|
+
"Require explicit registry approval before removing preset coverage."
|
|
2720
|
+
),
|
|
2721
|
+
"required_invariant_families_removed": (
|
|
2722
|
+
"Require explicit registry approval before removing invariant-family coverage."
|
|
2723
|
+
),
|
|
2724
|
+
"selected_patch_changed": (
|
|
2725
|
+
"Review the selected patch diff and rerun staging replay with the new "
|
|
2726
|
+
"patch signature."
|
|
2727
|
+
),
|
|
2728
|
+
"optimizer_backend_changed": (
|
|
2729
|
+
"Review backend-lineage evidence and confirm the new optimizer backend "
|
|
2730
|
+
"is expected for this replay pack."
|
|
2731
|
+
),
|
|
2732
|
+
"case_signature_changed": (
|
|
2733
|
+
"Inspect added or removed case ids and confirm selection coverage still "
|
|
2734
|
+
"represents required registry families."
|
|
2735
|
+
),
|
|
2736
|
+
"retention_key_changed": (
|
|
2737
|
+
"Record the retention-key change with the registry release metadata."
|
|
2738
|
+
),
|
|
2739
|
+
"promotion_status_changed": (
|
|
2740
|
+
"Record promotion-status drift and compare latest promotion failures."
|
|
2741
|
+
),
|
|
2742
|
+
"required_contract_expanded": (
|
|
2743
|
+
"Record the new required preset or invariant-family coverage in the "
|
|
2744
|
+
"registry release notes."
|
|
2745
|
+
),
|
|
2746
|
+
"coverage_drift": (
|
|
2747
|
+
"Record coverage drift alongside the retained Future AGI replay pack."
|
|
2748
|
+
),
|
|
2749
|
+
"optimizer_score_drift": (
|
|
2750
|
+
"Record optimizer-score drift and monitor the next replay run."
|
|
2751
|
+
),
|
|
2752
|
+
"missing_optimizer_score_delta": (
|
|
2753
|
+
"Attach promotion checks for adjacent lineage entries to compare optimizer scores."
|
|
2754
|
+
),
|
|
2755
|
+
}
|
|
2756
|
+
recommendations: list[str] = []
|
|
2757
|
+
for reason in list(blocking_reasons) + list(warnings):
|
|
2758
|
+
recommendation = recommendation_by_reason.get(reason)
|
|
2759
|
+
if recommendation:
|
|
2760
|
+
recommendations.append(recommendation)
|
|
2761
|
+
return _unique_strings(recommendations)
|
|
2762
|
+
|
|
2763
|
+
|
|
2764
|
+
def _optimizer_result_score(value: Any) -> Optional[float]:
|
|
2765
|
+
if value is None:
|
|
2766
|
+
return None
|
|
2767
|
+
if isinstance(value, Mapping):
|
|
2768
|
+
for key in ("final_score", "score", "optimizer_score"):
|
|
2769
|
+
if value.get(key) is not None:
|
|
2770
|
+
try:
|
|
2771
|
+
return float(value[key])
|
|
2772
|
+
except (TypeError, ValueError):
|
|
2773
|
+
return None
|
|
2774
|
+
nested = value.get("result") or value.get("reoptimization_result")
|
|
2775
|
+
if nested is not None:
|
|
2776
|
+
return _optimizer_result_score(nested)
|
|
2777
|
+
return None
|
|
2778
|
+
for key in ("final_score", "score", "optimizer_score"):
|
|
2779
|
+
item = getattr(value, key, None)
|
|
2780
|
+
if item is not None:
|
|
2781
|
+
try:
|
|
2782
|
+
return float(item)
|
|
2783
|
+
except (TypeError, ValueError):
|
|
2784
|
+
return None
|
|
2785
|
+
return None
|
|
2786
|
+
|
|
2787
|
+
|
|
2788
|
+
def _futureagi_experiment_payload(
|
|
2789
|
+
client: Any,
|
|
2790
|
+
*,
|
|
2791
|
+
experiment_id: str,
|
|
2792
|
+
page_size: int,
|
|
2793
|
+
max_pages: int,
|
|
2794
|
+
include_rows: bool,
|
|
2795
|
+
include_stats: bool,
|
|
2796
|
+
prefer_v2: bool,
|
|
2797
|
+
) -> dict[str, Any]:
|
|
2798
|
+
if _futureagi_experiment_payload_like(client):
|
|
2799
|
+
payload = _load_payload(client)
|
|
2800
|
+
return dict(payload) if isinstance(payload, Mapping) else {"records": payload}
|
|
2801
|
+
|
|
2802
|
+
method = getattr(client, "fetch_experiment_history", None)
|
|
2803
|
+
if not callable(method):
|
|
2804
|
+
raise TypeError(
|
|
2805
|
+
"client must expose fetch_experiment_history() or be a Future AGI "
|
|
2806
|
+
"experiment-history payload."
|
|
2807
|
+
)
|
|
2808
|
+
attempts = (
|
|
2809
|
+
lambda: method(
|
|
2810
|
+
experiment_id=experiment_id,
|
|
2811
|
+
page_size=page_size,
|
|
2812
|
+
max_pages=max_pages,
|
|
2813
|
+
include_rows=include_rows,
|
|
2814
|
+
include_stats=include_stats,
|
|
2815
|
+
prefer_v2=prefer_v2,
|
|
2816
|
+
),
|
|
2817
|
+
lambda: method(
|
|
2818
|
+
experiment_id=experiment_id,
|
|
2819
|
+
page_size=page_size,
|
|
2820
|
+
max_pages=max_pages,
|
|
2821
|
+
),
|
|
2822
|
+
lambda: method(experiment_id),
|
|
2823
|
+
)
|
|
2824
|
+
last_error: Optional[TypeError] = None
|
|
2825
|
+
for attempt in attempts:
|
|
2826
|
+
try:
|
|
2827
|
+
payload = _load_payload(attempt())
|
|
2828
|
+
return dict(payload) if isinstance(payload, Mapping) else {"records": payload}
|
|
2829
|
+
except TypeError as exc:
|
|
2830
|
+
last_error = exc
|
|
2831
|
+
if last_error is not None:
|
|
2832
|
+
raise last_error
|
|
2833
|
+
return {}
|
|
2834
|
+
|
|
2835
|
+
|
|
2836
|
+
def _futureagi_experiment_payload_like(value: Any) -> bool:
|
|
2837
|
+
if isinstance(value, Mapping):
|
|
2838
|
+
return any(
|
|
2839
|
+
key in value
|
|
2840
|
+
for key in (
|
|
2841
|
+
"experiment",
|
|
2842
|
+
"experiment_id",
|
|
2843
|
+
"detail",
|
|
2844
|
+
"stats",
|
|
2845
|
+
"rows",
|
|
2846
|
+
"records",
|
|
2847
|
+
"variants",
|
|
2848
|
+
"rankings",
|
|
2849
|
+
"results",
|
|
2850
|
+
"history",
|
|
2851
|
+
)
|
|
2852
|
+
)
|
|
2853
|
+
return False
|
|
2854
|
+
|
|
2855
|
+
|
|
2856
|
+
def _futureagi_experiment_metadata(
|
|
2857
|
+
payload: Mapping[str, Any],
|
|
2858
|
+
*,
|
|
2859
|
+
experiment_id: str,
|
|
2860
|
+
) -> dict[str, Any]:
|
|
2861
|
+
metadata: dict[str, Any] = {"id": experiment_id}
|
|
2862
|
+
for key in ("experiment", "detail"):
|
|
2863
|
+
section = payload.get(key)
|
|
2864
|
+
if not isinstance(section, Mapping):
|
|
2865
|
+
continue
|
|
2866
|
+
result = _ensure_mapping(_futureagi_payload_result(section))
|
|
2867
|
+
if not result:
|
|
2868
|
+
continue
|
|
2869
|
+
for field in ("id", "name", "status", "dataset", "dataset_id", "framework", "runtime"):
|
|
2870
|
+
value = result.get(field)
|
|
2871
|
+
if value is not None:
|
|
2872
|
+
metadata[field] = value
|
|
2873
|
+
for field in ("experiment_id", "experiment_name", "status", "framework", "runtime"):
|
|
2874
|
+
value = payload.get(field)
|
|
2875
|
+
if value is not None:
|
|
2876
|
+
metadata[field.replace("experiment_", "")] = value
|
|
2877
|
+
return metadata
|
|
2878
|
+
|
|
2879
|
+
|
|
2880
|
+
def _futureagi_experiment_observation_records(
|
|
2881
|
+
payload: Mapping[str, Any],
|
|
2882
|
+
*,
|
|
2883
|
+
experiment_id: str,
|
|
2884
|
+
experiment_metadata: Mapping[str, Any],
|
|
2885
|
+
) -> list[dict[str, Any]]:
|
|
2886
|
+
records: list[dict[str, Any]] = []
|
|
2887
|
+
records.extend(
|
|
2888
|
+
_futureagi_explicit_history_records(
|
|
2889
|
+
payload,
|
|
2890
|
+
experiment_id=experiment_id,
|
|
2891
|
+
experiment_metadata=experiment_metadata,
|
|
2892
|
+
)
|
|
2893
|
+
)
|
|
2894
|
+
records.extend(
|
|
2895
|
+
_futureagi_experiment_stats_records(
|
|
2896
|
+
payload,
|
|
2897
|
+
experiment_id=experiment_id,
|
|
2898
|
+
experiment_metadata=experiment_metadata,
|
|
2899
|
+
)
|
|
2900
|
+
)
|
|
2901
|
+
records.extend(
|
|
2902
|
+
_futureagi_experiment_row_records(
|
|
2903
|
+
payload,
|
|
2904
|
+
experiment_id=experiment_id,
|
|
2905
|
+
experiment_metadata=experiment_metadata,
|
|
2906
|
+
)
|
|
2907
|
+
)
|
|
2908
|
+
return _dedupe_futureagi_observability_records(records)
|
|
2909
|
+
|
|
2910
|
+
|
|
2911
|
+
def _futureagi_explicit_history_records(
|
|
2912
|
+
payload: Mapping[str, Any],
|
|
2913
|
+
*,
|
|
2914
|
+
experiment_id: str,
|
|
2915
|
+
experiment_metadata: Mapping[str, Any],
|
|
2916
|
+
) -> list[dict[str, Any]]:
|
|
2917
|
+
records: list[dict[str, Any]] = []
|
|
2918
|
+
for key in ("records", "history", "observations"):
|
|
2919
|
+
value = payload.get(key)
|
|
2920
|
+
if not isinstance(value, Sequence) or isinstance(value, (str, bytes, bytearray)):
|
|
2921
|
+
continue
|
|
2922
|
+
for index, item in enumerate(value, start=1):
|
|
2923
|
+
if not isinstance(item, Mapping):
|
|
2924
|
+
continue
|
|
2925
|
+
record = dict(item)
|
|
2926
|
+
record.setdefault("id", f"{experiment_id}:history:{index}")
|
|
2927
|
+
records.append(
|
|
2928
|
+
_futureagi_experiment_record(
|
|
2929
|
+
record,
|
|
2930
|
+
experiment_id=experiment_id,
|
|
2931
|
+
experiment_metadata=experiment_metadata,
|
|
2932
|
+
source_section=key,
|
|
2933
|
+
index=index,
|
|
2934
|
+
)
|
|
2935
|
+
)
|
|
2936
|
+
return records
|
|
2937
|
+
|
|
2938
|
+
|
|
2939
|
+
def _futureagi_experiment_stats_records(
|
|
2940
|
+
payload: Mapping[str, Any],
|
|
2941
|
+
*,
|
|
2942
|
+
experiment_id: str,
|
|
2943
|
+
experiment_metadata: Mapping[str, Any],
|
|
2944
|
+
) -> list[dict[str, Any]]:
|
|
2945
|
+
sections: list[tuple[str, Any]] = []
|
|
2946
|
+
for key in ("stats", "results", "comparisons", "list"):
|
|
2947
|
+
value = payload.get(key)
|
|
2948
|
+
if value is not None:
|
|
2949
|
+
sections.append((key, value))
|
|
2950
|
+
if payload.get("variants") is not None or payload.get("rankings") is not None:
|
|
2951
|
+
sections.append(("payload", payload))
|
|
2952
|
+
|
|
2953
|
+
records: list[dict[str, Any]] = []
|
|
2954
|
+
for section_name, section in sections:
|
|
2955
|
+
result = _futureagi_payload_result(section)
|
|
2956
|
+
candidates = _futureagi_variant_rows(result)
|
|
2957
|
+
if section_name == "list":
|
|
2958
|
+
matching = [
|
|
2959
|
+
row
|
|
2960
|
+
for row in candidates
|
|
2961
|
+
if isinstance(row, Mapping)
|
|
2962
|
+
and str(row.get("id") or row.get("experiment_id") or "") == experiment_id
|
|
2963
|
+
]
|
|
2964
|
+
candidates = matching or candidates
|
|
2965
|
+
for index, row in enumerate(candidates, start=1):
|
|
2966
|
+
if not isinstance(row, Mapping):
|
|
2967
|
+
continue
|
|
2968
|
+
metrics = _futureagi_metrics_from_mapping(row)
|
|
2969
|
+
if not metrics:
|
|
2970
|
+
metrics = _futureagi_status_metrics_from_mapping(row)
|
|
2971
|
+
if not metrics:
|
|
2972
|
+
continue
|
|
2973
|
+
candidate_id = _futureagi_variant_id(row, fallback=f"variant-{index}")
|
|
2974
|
+
record = {
|
|
2975
|
+
"id": f"{experiment_id}:variant:{candidate_id}",
|
|
2976
|
+
"run_id": f"{experiment_id}:variant:{candidate_id}",
|
|
2977
|
+
"candidate_id": str(candidate_id),
|
|
2978
|
+
"metrics": metrics,
|
|
2979
|
+
"score": _futureagi_record_score(metrics, row),
|
|
2980
|
+
"status": row.get("status") or row.get("state"),
|
|
2981
|
+
"raw_variant": dict(row),
|
|
2982
|
+
}
|
|
2983
|
+
records.append(
|
|
2984
|
+
_futureagi_experiment_record(
|
|
2985
|
+
record,
|
|
2986
|
+
experiment_id=experiment_id,
|
|
2987
|
+
experiment_metadata=experiment_metadata,
|
|
2988
|
+
source_section="stats",
|
|
2989
|
+
index=index,
|
|
2990
|
+
)
|
|
2991
|
+
)
|
|
2992
|
+
return records
|
|
2993
|
+
|
|
2994
|
+
|
|
2995
|
+
def _futureagi_experiment_row_records(
|
|
2996
|
+
payload: Mapping[str, Any],
|
|
2997
|
+
*,
|
|
2998
|
+
experiment_id: str,
|
|
2999
|
+
experiment_metadata: Mapping[str, Any],
|
|
3000
|
+
) -> list[dict[str, Any]]:
|
|
3001
|
+
raw_pages = payload.get("rows")
|
|
3002
|
+
if raw_pages is None:
|
|
3003
|
+
raw_pages = payload.get("row_pages")
|
|
3004
|
+
if raw_pages is None and (
|
|
3005
|
+
"table" in payload or "column_config" in payload or "columnConfig" in payload
|
|
3006
|
+
):
|
|
3007
|
+
raw_pages = [payload]
|
|
3008
|
+
if isinstance(raw_pages, Mapping):
|
|
3009
|
+
raw_pages = [raw_pages]
|
|
3010
|
+
if not isinstance(raw_pages, Sequence) or isinstance(raw_pages, (str, bytes, bytearray)):
|
|
3011
|
+
return []
|
|
3012
|
+
|
|
3013
|
+
records: list[dict[str, Any]] = []
|
|
3014
|
+
for page_index, raw_page in enumerate(raw_pages):
|
|
3015
|
+
result = _futureagi_payload_result(raw_page)
|
|
3016
|
+
columns = _futureagi_table_columns(result)
|
|
3017
|
+
rows = _futureagi_table_rows(result)
|
|
3018
|
+
for row_index, row in enumerate(rows, start=1):
|
|
3019
|
+
values, row_metadata = _futureagi_row_values(row, columns=columns)
|
|
3020
|
+
metrics = _futureagi_metrics_from_row_values(values, columns=columns)
|
|
3021
|
+
if not metrics:
|
|
3022
|
+
continue
|
|
3023
|
+
row_id = row_metadata.get("row_id") or f"page-{page_index}-row-{row_index}"
|
|
3024
|
+
record = {
|
|
3025
|
+
"id": f"{experiment_id}:row:{row_id}",
|
|
3026
|
+
"run_id": f"{experiment_id}:row:{row_id}",
|
|
3027
|
+
"candidate_id": _futureagi_row_candidate_id(values, row),
|
|
3028
|
+
"metrics": metrics,
|
|
3029
|
+
"score": _futureagi_record_score(metrics, values),
|
|
3030
|
+
"raw_row": copy.deepcopy(row),
|
|
3031
|
+
"row_values": copy.deepcopy(values),
|
|
3032
|
+
"futureagi_row_id": row_id,
|
|
3033
|
+
"futureagi_row_order": row_metadata.get("order"),
|
|
3034
|
+
}
|
|
3035
|
+
records.append(
|
|
3036
|
+
_futureagi_experiment_record(
|
|
3037
|
+
record,
|
|
3038
|
+
experiment_id=experiment_id,
|
|
3039
|
+
experiment_metadata=experiment_metadata,
|
|
3040
|
+
source_section="rows",
|
|
3041
|
+
index=len(records) + 1,
|
|
3042
|
+
)
|
|
3043
|
+
)
|
|
3044
|
+
return records
|
|
3045
|
+
|
|
3046
|
+
|
|
3047
|
+
def _futureagi_experiment_record(
|
|
3048
|
+
record: Mapping[str, Any],
|
|
3049
|
+
*,
|
|
3050
|
+
experiment_id: str,
|
|
3051
|
+
experiment_metadata: Mapping[str, Any],
|
|
3052
|
+
source_section: str,
|
|
3053
|
+
index: int,
|
|
3054
|
+
) -> dict[str, Any]:
|
|
3055
|
+
normalized = dict(record)
|
|
3056
|
+
normalized.setdefault("source", "futureagi")
|
|
3057
|
+
normalized.setdefault("framework", experiment_metadata.get("framework") or "generic")
|
|
3058
|
+
normalized.setdefault("run_id", normalized.get("id") or f"{experiment_id}:{source_section}:{index}")
|
|
3059
|
+
normalized.setdefault("experiment_id", experiment_id)
|
|
3060
|
+
normalized.setdefault("experiment_name", experiment_metadata.get("name"))
|
|
3061
|
+
metadata = _ensure_mapping(normalized.get("metadata"))
|
|
3062
|
+
metadata.update(
|
|
3063
|
+
{
|
|
3064
|
+
"kind": "futureagi_experiment_history_record",
|
|
3065
|
+
"futureagi_experiment_id": experiment_id,
|
|
3066
|
+
"futureagi_experiment_name": experiment_metadata.get("name"),
|
|
3067
|
+
"futureagi_source_section": source_section,
|
|
3068
|
+
"futureagi_record_index": index,
|
|
3069
|
+
}
|
|
3070
|
+
)
|
|
3071
|
+
normalized["metadata"] = metadata
|
|
3072
|
+
return normalized
|
|
3073
|
+
|
|
3074
|
+
|
|
3075
|
+
def _futureagi_variant_rows(result: Any) -> list[Any]:
|
|
3076
|
+
if isinstance(result, Mapping):
|
|
3077
|
+
for key in ("table_data", "tableData", "variants", "rankings", "comparisons", "results"):
|
|
3078
|
+
value = result.get(key)
|
|
3079
|
+
if isinstance(value, Sequence) and not isinstance(value, (str, bytes, bytearray)):
|
|
3080
|
+
return list(value)
|
|
3081
|
+
table_rows = _futureagi_table_rows(result)
|
|
3082
|
+
if table_rows:
|
|
3083
|
+
return table_rows
|
|
3084
|
+
if isinstance(result, Sequence) and not isinstance(result, (str, bytes, bytearray)):
|
|
3085
|
+
return list(result)
|
|
3086
|
+
return []
|
|
3087
|
+
|
|
3088
|
+
|
|
3089
|
+
def _futureagi_metrics_from_row_values(
|
|
3090
|
+
values: Mapping[str, Any],
|
|
3091
|
+
*,
|
|
3092
|
+
columns: Sequence[Mapping[str, Any]],
|
|
3093
|
+
) -> dict[str, float]:
|
|
3094
|
+
columns_by_name = {str(column.get("name") or ""): column for column in columns}
|
|
3095
|
+
metrics: dict[str, float] = {}
|
|
3096
|
+
for name, value in values.items():
|
|
3097
|
+
column = columns_by_name.get(str(name), {})
|
|
3098
|
+
if not _futureagi_column_looks_like_metric(name, column):
|
|
3099
|
+
continue
|
|
3100
|
+
score = _futureagi_metric_score(value)
|
|
3101
|
+
if score is not None:
|
|
3102
|
+
metrics[_normalize_metric_name(name)] = score
|
|
3103
|
+
return metrics
|
|
3104
|
+
|
|
3105
|
+
|
|
3106
|
+
def _futureagi_metrics_from_mapping(row: Mapping[str, Any]) -> dict[str, float]:
|
|
3107
|
+
metrics: dict[str, float] = {}
|
|
3108
|
+
nested_metrics = row.get("metrics") or row.get("metric_averages") or row.get("scores")
|
|
3109
|
+
if isinstance(nested_metrics, Mapping):
|
|
3110
|
+
for name, value in nested_metrics.items():
|
|
3111
|
+
score = _futureagi_metric_score(value)
|
|
3112
|
+
if score is not None:
|
|
3113
|
+
metrics[_normalize_metric_name(name)] = score
|
|
3114
|
+
for name, value in row.items():
|
|
3115
|
+
if not _futureagi_key_looks_like_metric(name):
|
|
3116
|
+
continue
|
|
3117
|
+
score = _futureagi_metric_score(value)
|
|
3118
|
+
if score is not None:
|
|
3119
|
+
metrics[_normalize_metric_name(name)] = score
|
|
3120
|
+
return metrics
|
|
3121
|
+
|
|
3122
|
+
|
|
3123
|
+
def _futureagi_status_metrics_from_mapping(row: Mapping[str, Any]) -> dict[str, float]:
|
|
3124
|
+
status = str(row.get("status") or row.get("state") or "").strip().lower()
|
|
3125
|
+
if not status:
|
|
3126
|
+
return {}
|
|
3127
|
+
completed = {
|
|
3128
|
+
"completed",
|
|
3129
|
+
"complete",
|
|
3130
|
+
"success",
|
|
3131
|
+
"succeeded",
|
|
3132
|
+
"passed",
|
|
3133
|
+
"pass",
|
|
3134
|
+
"done",
|
|
3135
|
+
}
|
|
3136
|
+
return {"experiment_completed": 1.0 if status in completed else 0.0}
|
|
3137
|
+
|
|
3138
|
+
|
|
3139
|
+
def _futureagi_metric_score(value: Any) -> Optional[float]:
|
|
3140
|
+
value = _futureagi_cell_payload_value(value)
|
|
3141
|
+
if isinstance(value, Mapping):
|
|
3142
|
+
for key in ("score", "value", "output", "cell_value", "cellValue", "average"):
|
|
3143
|
+
score = _futureagi_metric_score(value.get(key))
|
|
3144
|
+
if score is not None:
|
|
3145
|
+
return score
|
|
3146
|
+
return None
|
|
3147
|
+
if isinstance(value, str):
|
|
3148
|
+
stripped = value.strip()
|
|
3149
|
+
if not stripped:
|
|
3150
|
+
return None
|
|
3151
|
+
try:
|
|
3152
|
+
value = float(stripped.rstrip("%"))
|
|
3153
|
+
except ValueError:
|
|
3154
|
+
return None
|
|
3155
|
+
if stripped.endswith("%"):
|
|
3156
|
+
value = value / 100.0
|
|
3157
|
+
if isinstance(value, bool):
|
|
3158
|
+
return 1.0 if value else 0.0
|
|
3159
|
+
if isinstance(value, (int, float)):
|
|
3160
|
+
numeric = float(value)
|
|
3161
|
+
if numeric > 1.0 and numeric <= 100.0:
|
|
3162
|
+
numeric = numeric / 100.0
|
|
3163
|
+
return max(0.0, min(numeric, 1.0))
|
|
3164
|
+
return None
|
|
3165
|
+
|
|
3166
|
+
|
|
3167
|
+
def _futureagi_record_score(
|
|
3168
|
+
metrics: Mapping[str, float],
|
|
3169
|
+
raw: Mapping[str, Any],
|
|
3170
|
+
) -> float:
|
|
3171
|
+
explicit = _futureagi_metric_score(
|
|
3172
|
+
raw.get("score")
|
|
3173
|
+
or raw.get("avg_score")
|
|
3174
|
+
or raw.get("average_score")
|
|
3175
|
+
or raw.get("overall_rating")
|
|
3176
|
+
or raw.get("overall")
|
|
3177
|
+
)
|
|
3178
|
+
if explicit is not None:
|
|
3179
|
+
return explicit
|
|
3180
|
+
return sum(metrics.values()) / len(metrics) if metrics else 1.0
|
|
3181
|
+
|
|
3182
|
+
|
|
3183
|
+
def _futureagi_column_looks_like_metric(name: str, column: Mapping[str, Any]) -> bool:
|
|
3184
|
+
source = str(
|
|
3185
|
+
column.get("source")
|
|
3186
|
+
or column.get("origin_type")
|
|
3187
|
+
or column.get("originType")
|
|
3188
|
+
or column.get("type")
|
|
3189
|
+
or ""
|
|
3190
|
+
).lower()
|
|
3191
|
+
if "evaluation" in source or "eval" in source or "score" in source:
|
|
3192
|
+
return True
|
|
3193
|
+
return _futureagi_key_looks_like_metric(name)
|
|
3194
|
+
|
|
3195
|
+
|
|
3196
|
+
def _futureagi_key_looks_like_metric(name: Any) -> bool:
|
|
3197
|
+
normalized = _normalize_metric_name(name)
|
|
3198
|
+
if not normalized:
|
|
3199
|
+
return False
|
|
3200
|
+
excluded_tokens = (
|
|
3201
|
+
"id",
|
|
3202
|
+
"name",
|
|
3203
|
+
"dataset",
|
|
3204
|
+
"variant",
|
|
3205
|
+
"tokens",
|
|
3206
|
+
"token",
|
|
3207
|
+
"latency",
|
|
3208
|
+
"response_time",
|
|
3209
|
+
"duration",
|
|
3210
|
+
"runtime",
|
|
3211
|
+
"status",
|
|
3212
|
+
"order",
|
|
3213
|
+
"rank",
|
|
3214
|
+
"created",
|
|
3215
|
+
"updated",
|
|
3216
|
+
)
|
|
3217
|
+
if normalized in excluded_tokens or any(token == normalized for token in excluded_tokens):
|
|
3218
|
+
return False
|
|
3219
|
+
metric_tokens = (
|
|
3220
|
+
"score",
|
|
3221
|
+
"quality",
|
|
3222
|
+
"accuracy",
|
|
3223
|
+
"adherence",
|
|
3224
|
+
"outcome",
|
|
3225
|
+
"success",
|
|
3226
|
+
"safety",
|
|
3227
|
+
"correctness",
|
|
3228
|
+
"coverage",
|
|
3229
|
+
"grounding",
|
|
3230
|
+
"resilience",
|
|
3231
|
+
"coordination",
|
|
3232
|
+
"memory",
|
|
3233
|
+
"tool",
|
|
3234
|
+
"policy",
|
|
3235
|
+
)
|
|
3236
|
+
return any(token in normalized for token in metric_tokens)
|
|
3237
|
+
|
|
3238
|
+
|
|
3239
|
+
def _normalize_metric_name(name: Any) -> str:
|
|
3240
|
+
return _case_slug(name).replace("-", "_")
|
|
3241
|
+
|
|
3242
|
+
|
|
3243
|
+
def _futureagi_variant_id(row: Mapping[str, Any], *, fallback: str) -> str:
|
|
3244
|
+
for key in (
|
|
3245
|
+
"candidate_id",
|
|
3246
|
+
"variant_id",
|
|
3247
|
+
"dataset_id",
|
|
3248
|
+
"id",
|
|
3249
|
+
"experiment_dataset_id",
|
|
3250
|
+
"experiment_dataset_name",
|
|
3251
|
+
"variant",
|
|
3252
|
+
"name",
|
|
3253
|
+
):
|
|
3254
|
+
value = row.get(key)
|
|
3255
|
+
if value is not None:
|
|
3256
|
+
return _case_slug(value) or str(value)
|
|
3257
|
+
return fallback
|
|
3258
|
+
|
|
3259
|
+
|
|
3260
|
+
def _futureagi_row_candidate_id(values: Mapping[str, Any], row: Any) -> Optional[str]:
|
|
3261
|
+
for source in (values, row if isinstance(row, Mapping) else {}):
|
|
3262
|
+
if not isinstance(source, Mapping):
|
|
3263
|
+
continue
|
|
3264
|
+
for key in ("candidate_id", "variant_id", "experiment_dataset_id", "dataset_id"):
|
|
3265
|
+
value = source.get(key)
|
|
3266
|
+
if value is not None:
|
|
3267
|
+
return str(value)
|
|
3268
|
+
return None
|
|
3269
|
+
|
|
3270
|
+
|
|
3271
|
+
def _dedupe_futureagi_observability_records(
|
|
3272
|
+
records: Sequence[Mapping[str, Any]],
|
|
3273
|
+
) -> list[dict[str, Any]]:
|
|
3274
|
+
deduped: dict[str, dict[str, Any]] = {}
|
|
3275
|
+
for index, record in enumerate(records):
|
|
3276
|
+
key = str(record.get("id") or record.get("run_id") or index)
|
|
3277
|
+
deduped[key] = dict(record)
|
|
3278
|
+
return list(deduped.values())
|
|
3279
|
+
|
|
3280
|
+
|
|
3281
|
+
def _penalize_missing_futureagi_experiment_metrics(
|
|
3282
|
+
records: Sequence[AgentObservabilityRecord],
|
|
3283
|
+
required_metrics: Mapping[str, float],
|
|
3284
|
+
) -> None:
|
|
3285
|
+
if not required_metrics:
|
|
3286
|
+
return
|
|
3287
|
+
required = set(required_metrics)
|
|
3288
|
+
for record in records:
|
|
3289
|
+
if record.passed or required & set(record.metrics):
|
|
3290
|
+
continue
|
|
3291
|
+
record.score = 0.0
|
|
3292
|
+
|
|
3293
|
+
|
|
3294
|
+
def _futureagi_dataset_payloads(
|
|
3295
|
+
client: Any,
|
|
3296
|
+
*,
|
|
3297
|
+
dataset_id: str,
|
|
3298
|
+
page_size: int,
|
|
3299
|
+
max_pages: int,
|
|
3300
|
+
) -> list[Any]:
|
|
3301
|
+
if _futureagi_payload_like(client):
|
|
3302
|
+
return [client]
|
|
3303
|
+
if isinstance(client, Sequence) and not isinstance(client, (str, bytes, bytearray)):
|
|
3304
|
+
if all(_futureagi_payload_like(item) for item in client):
|
|
3305
|
+
return list(client)
|
|
3306
|
+
|
|
3307
|
+
payloads: list[Any] = []
|
|
3308
|
+
for page_index in range(max_pages):
|
|
3309
|
+
payload = _fetch_futureagi_dataset_page(
|
|
3310
|
+
client,
|
|
3311
|
+
dataset_id=dataset_id,
|
|
3312
|
+
page_size=page_size,
|
|
3313
|
+
current_page_index=page_index,
|
|
3314
|
+
)
|
|
3315
|
+
payloads.append(payload)
|
|
3316
|
+
result = _futureagi_payload_result(payload)
|
|
3317
|
+
total_pages = _futureagi_total_pages(result)
|
|
3318
|
+
row_count = len(_futureagi_table_rows(result))
|
|
3319
|
+
if total_pages is not None:
|
|
3320
|
+
if page_index + 1 >= total_pages:
|
|
3321
|
+
break
|
|
3322
|
+
elif row_count < page_size:
|
|
3323
|
+
break
|
|
3324
|
+
else:
|
|
3325
|
+
break
|
|
3326
|
+
return payloads
|
|
3327
|
+
|
|
3328
|
+
|
|
3329
|
+
def _futureagi_payload_like(value: Any) -> bool:
|
|
3330
|
+
if isinstance(value, Mapping):
|
|
3331
|
+
return any(
|
|
3332
|
+
key in value
|
|
3333
|
+
for key in (
|
|
3334
|
+
"result",
|
|
3335
|
+
"table",
|
|
3336
|
+
"rows",
|
|
3337
|
+
"column_config",
|
|
3338
|
+
"columnConfig",
|
|
3339
|
+
"columns",
|
|
3340
|
+
)
|
|
3341
|
+
)
|
|
3342
|
+
return hasattr(value, "rows") and hasattr(value, "columns")
|
|
3343
|
+
|
|
3344
|
+
|
|
3345
|
+
def _fetch_futureagi_dataset_page(
|
|
3346
|
+
client: Any,
|
|
3347
|
+
*,
|
|
3348
|
+
dataset_id: str,
|
|
3349
|
+
page_size: int,
|
|
3350
|
+
current_page_index: int,
|
|
3351
|
+
) -> Any:
|
|
3352
|
+
for method_name in (
|
|
3353
|
+
"fetch_regression_dataset",
|
|
3354
|
+
"fetch_dataset_table",
|
|
3355
|
+
"get_dataset_table",
|
|
3356
|
+
"fetch_dataset",
|
|
3357
|
+
"get_dataset",
|
|
3358
|
+
):
|
|
3359
|
+
method = getattr(client, method_name, None)
|
|
3360
|
+
if not callable(method):
|
|
3361
|
+
continue
|
|
3362
|
+
attempts = (
|
|
3363
|
+
lambda: method(
|
|
3364
|
+
dataset_id=dataset_id,
|
|
3365
|
+
page_size=page_size,
|
|
3366
|
+
current_page_index=current_page_index,
|
|
3367
|
+
),
|
|
3368
|
+
lambda: method(
|
|
3369
|
+
dataset_id=dataset_id,
|
|
3370
|
+
page_size=page_size,
|
|
3371
|
+
page_index=current_page_index,
|
|
3372
|
+
),
|
|
3373
|
+
lambda: method(
|
|
3374
|
+
dataset_id,
|
|
3375
|
+
page_size=page_size,
|
|
3376
|
+
current_page_index=current_page_index,
|
|
3377
|
+
),
|
|
3378
|
+
lambda: method(dataset_id),
|
|
3379
|
+
)
|
|
3380
|
+
last_error: Optional[TypeError] = None
|
|
3381
|
+
for attempt in attempts:
|
|
3382
|
+
try:
|
|
3383
|
+
return attempt()
|
|
3384
|
+
except TypeError as exc:
|
|
3385
|
+
last_error = exc
|
|
3386
|
+
if last_error is not None:
|
|
3387
|
+
raise last_error
|
|
3388
|
+
|
|
3389
|
+
raise TypeError(
|
|
3390
|
+
"client must expose fetch_regression_dataset(), fetch_dataset_table(), "
|
|
3391
|
+
"get_dataset_table(), fetch_dataset(), or get_dataset()."
|
|
3392
|
+
)
|
|
3393
|
+
|
|
3394
|
+
|
|
3395
|
+
def _futureagi_regression_cases_from_payloads(
|
|
3396
|
+
payloads: Sequence[Any],
|
|
3397
|
+
*,
|
|
3398
|
+
dataset_id: str,
|
|
3399
|
+
) -> tuple[list[AgentRegressionCase], dict[str, Any]]:
|
|
3400
|
+
cases: list[AgentRegressionCase] = []
|
|
3401
|
+
table_metadata: dict[str, Any] = {}
|
|
3402
|
+
max_column_count = 0
|
|
3403
|
+
for payload in payloads:
|
|
3404
|
+
result = _futureagi_payload_result(payload)
|
|
3405
|
+
columns = _futureagi_table_columns(result)
|
|
3406
|
+
max_column_count = max(max_column_count, len(columns))
|
|
3407
|
+
page_metadata = _futureagi_table_metadata(result)
|
|
3408
|
+
for key, value in page_metadata.items():
|
|
3409
|
+
if value is not None:
|
|
3410
|
+
table_metadata[key] = value
|
|
3411
|
+
for row in _futureagi_table_rows(result):
|
|
3412
|
+
cases.append(
|
|
3413
|
+
_futureagi_regression_case_from_row(
|
|
3414
|
+
row,
|
|
3415
|
+
columns=columns,
|
|
3416
|
+
dataset_id=dataset_id,
|
|
3417
|
+
index=len(cases) + 1,
|
|
3418
|
+
)
|
|
3419
|
+
)
|
|
3420
|
+
table_metadata["column_count"] = max_column_count
|
|
3421
|
+
return cases, table_metadata
|
|
3422
|
+
|
|
3423
|
+
|
|
3424
|
+
def _futureagi_payload_result(payload: Any) -> Any:
|
|
3425
|
+
if hasattr(payload, "model_dump"):
|
|
3426
|
+
payload = payload.model_dump()
|
|
3427
|
+
elif hasattr(payload, "dict"):
|
|
3428
|
+
payload = payload.dict()
|
|
3429
|
+
if isinstance(payload, Mapping):
|
|
3430
|
+
for key in ("result", "data", "dataset_table", "datasetTable"):
|
|
3431
|
+
value = payload.get(key)
|
|
3432
|
+
if value is not None:
|
|
3433
|
+
return value
|
|
3434
|
+
return payload
|
|
3435
|
+
return payload
|
|
3436
|
+
|
|
3437
|
+
|
|
3438
|
+
def _futureagi_table_columns(result: Any) -> list[dict[str, Any]]:
|
|
3439
|
+
raw_columns = _futureagi_get(
|
|
3440
|
+
result,
|
|
3441
|
+
"column_config",
|
|
3442
|
+
"columnConfig",
|
|
3443
|
+
"columns",
|
|
3444
|
+
)
|
|
3445
|
+
columns: list[dict[str, Any]] = []
|
|
3446
|
+
if isinstance(raw_columns, Sequence) and not isinstance(
|
|
3447
|
+
raw_columns,
|
|
3448
|
+
(str, bytes, bytearray),
|
|
3449
|
+
):
|
|
3450
|
+
for raw_column in raw_columns:
|
|
3451
|
+
column_id = _futureagi_scalar(
|
|
3452
|
+
_futureagi_get(raw_column, "id", "column_id", "columnId")
|
|
3453
|
+
)
|
|
3454
|
+
name = _futureagi_scalar(
|
|
3455
|
+
_futureagi_get(raw_column, "name", "column_name", "columnName")
|
|
3456
|
+
)
|
|
3457
|
+
data_type = _futureagi_scalar(
|
|
3458
|
+
_futureagi_get(raw_column, "data_type", "dataType", "type")
|
|
3459
|
+
)
|
|
3460
|
+
if not name and column_id:
|
|
3461
|
+
name = column_id
|
|
3462
|
+
if name:
|
|
3463
|
+
columns.append(
|
|
3464
|
+
{
|
|
3465
|
+
"id": column_id or name,
|
|
3466
|
+
"name": str(name),
|
|
3467
|
+
"data_type": str(data_type or "text").lower(),
|
|
3468
|
+
}
|
|
3469
|
+
)
|
|
3470
|
+
if columns:
|
|
3471
|
+
return columns
|
|
3472
|
+
return [
|
|
3473
|
+
{"id": column["name"], "name": column["name"], "data_type": column["data_type"]}
|
|
3474
|
+
for column in _futureagi_dataset_columns()
|
|
3475
|
+
]
|
|
3476
|
+
|
|
3477
|
+
|
|
3478
|
+
def _futureagi_table_rows(result: Any) -> list[Any]:
|
|
3479
|
+
rows = _futureagi_get(result, "table", "rows")
|
|
3480
|
+
if isinstance(rows, Sequence) and not isinstance(rows, (str, bytes, bytearray)):
|
|
3481
|
+
return list(rows)
|
|
3482
|
+
return []
|
|
3483
|
+
|
|
3484
|
+
|
|
3485
|
+
def _futureagi_table_metadata(result: Any) -> dict[str, Any]:
|
|
3486
|
+
metadata = _futureagi_get(result, "metadata")
|
|
3487
|
+
if hasattr(metadata, "model_dump"):
|
|
3488
|
+
metadata = metadata.model_dump()
|
|
3489
|
+
elif hasattr(metadata, "dict"):
|
|
3490
|
+
metadata = metadata.dict()
|
|
3491
|
+
if isinstance(metadata, Mapping):
|
|
3492
|
+
return dict(metadata)
|
|
3493
|
+
return {}
|
|
3494
|
+
|
|
3495
|
+
|
|
3496
|
+
def _futureagi_total_pages(result: Any) -> Optional[int]:
|
|
3497
|
+
metadata = _futureagi_table_metadata(result)
|
|
3498
|
+
value = (
|
|
3499
|
+
metadata.get("total_pages")
|
|
3500
|
+
if "total_pages" in metadata
|
|
3501
|
+
else metadata.get("totalPages")
|
|
3502
|
+
)
|
|
3503
|
+
try:
|
|
3504
|
+
return int(value) if value is not None else None
|
|
3505
|
+
except (TypeError, ValueError):
|
|
3506
|
+
return None
|
|
3507
|
+
|
|
3508
|
+
|
|
3509
|
+
def _futureagi_regression_case_from_row(
|
|
3510
|
+
row: Any,
|
|
3511
|
+
*,
|
|
3512
|
+
columns: Sequence[Mapping[str, Any]],
|
|
3513
|
+
dataset_id: str,
|
|
3514
|
+
index: int,
|
|
3515
|
+
) -> AgentRegressionCase:
|
|
3516
|
+
values, row_metadata = _futureagi_row_values(row, columns=columns)
|
|
3517
|
+
case_id = str(
|
|
3518
|
+
values.get("case_id")
|
|
3519
|
+
or row_metadata.get("row_id")
|
|
3520
|
+
or f"futureagi-regression-row-{index}"
|
|
3521
|
+
)
|
|
3522
|
+
observability = _ensure_mapping(values.get("observability"))
|
|
3523
|
+
if not observability:
|
|
3524
|
+
failures = _string_list(values.get("response"))
|
|
3525
|
+
observability = {
|
|
3526
|
+
"source": "futureagi",
|
|
3527
|
+
"framework": "generic",
|
|
3528
|
+
"run_id": case_id,
|
|
3529
|
+
"score": 0.0 if failures else 1.0,
|
|
3530
|
+
"passed": not failures,
|
|
3531
|
+
"failures": failures,
|
|
3532
|
+
"metrics": {},
|
|
3533
|
+
"trace_signals": [],
|
|
3534
|
+
}
|
|
3535
|
+
expected = _ensure_mapping(values.get("expected_response"))
|
|
3536
|
+
if not expected:
|
|
3537
|
+
expected = {
|
|
3538
|
+
"should_pass": True,
|
|
3539
|
+
"required_metrics": {},
|
|
3540
|
+
"required_trace_signals": [],
|
|
3541
|
+
"previous_score": _coerce_score(observability.get("score")),
|
|
3542
|
+
"previous_failures": _string_list(observability.get("failures")),
|
|
3543
|
+
}
|
|
3544
|
+
tags = _string_list(values.get("tags"))
|
|
3545
|
+
case_metadata = _ensure_mapping(values.get("metadata"))
|
|
3546
|
+
row_id = row_metadata.get("row_id")
|
|
3547
|
+
order = row_metadata.get("order")
|
|
3548
|
+
query = values.get("query")
|
|
3549
|
+
response = values.get("response")
|
|
3550
|
+
case_metadata.update(
|
|
3551
|
+
{
|
|
3552
|
+
"kind": "futureagi_regression_case",
|
|
3553
|
+
"futureagi_dataset_id": dataset_id,
|
|
3554
|
+
"futureagi_row_id": row_id,
|
|
3555
|
+
"futureagi_order": order,
|
|
3556
|
+
"futureagi_query": query,
|
|
3557
|
+
"futureagi_response": response,
|
|
3558
|
+
}
|
|
3559
|
+
)
|
|
3560
|
+
case_metadata.setdefault("dataset_case_id", case_id)
|
|
3561
|
+
if isinstance(observability, Mapping):
|
|
3562
|
+
case_metadata.setdefault("source", observability.get("source"))
|
|
3563
|
+
case_metadata.setdefault("framework", observability.get("framework"))
|
|
3564
|
+
case_metadata.setdefault("run_id", observability.get("run_id"))
|
|
3565
|
+
case_metadata.setdefault("candidate_id", observability.get("candidate_id"))
|
|
3566
|
+
|
|
3567
|
+
return AgentRegressionCase(
|
|
3568
|
+
id=case_id,
|
|
3569
|
+
input={"observability": copy.deepcopy(dict(observability))},
|
|
3570
|
+
expected=copy.deepcopy(dict(expected)),
|
|
3571
|
+
tags=tags,
|
|
3572
|
+
metadata={
|
|
3573
|
+
key: value
|
|
3574
|
+
for key, value in case_metadata.items()
|
|
3575
|
+
if value is not None
|
|
3576
|
+
},
|
|
3577
|
+
)
|
|
3578
|
+
|
|
3579
|
+
|
|
3580
|
+
def _futureagi_row_values(
|
|
3581
|
+
row: Any,
|
|
3582
|
+
*,
|
|
3583
|
+
columns: Sequence[Mapping[str, Any]],
|
|
3584
|
+
) -> tuple[dict[str, Any], dict[str, Any]]:
|
|
3585
|
+
id_to_column = {
|
|
3586
|
+
str(column.get("id")): column
|
|
3587
|
+
for column in columns
|
|
3588
|
+
if column.get("id")
|
|
3589
|
+
}
|
|
3590
|
+
name_to_column = {
|
|
3591
|
+
str(column.get("name")): column
|
|
3592
|
+
for column in columns
|
|
3593
|
+
if column.get("name")
|
|
3594
|
+
}
|
|
3595
|
+
metadata = {
|
|
3596
|
+
"row_id": _futureagi_scalar(_futureagi_get(row, "row_id", "rowId", "id")),
|
|
3597
|
+
"order": _futureagi_get(row, "order"),
|
|
3598
|
+
}
|
|
3599
|
+
values: dict[str, Any] = {}
|
|
3600
|
+
|
|
3601
|
+
cells = _futureagi_get(row, "cells")
|
|
3602
|
+
if isinstance(cells, Sequence) and not isinstance(cells, (str, bytes, bytearray)):
|
|
3603
|
+
for cell in cells:
|
|
3604
|
+
column_id = _futureagi_scalar(
|
|
3605
|
+
_futureagi_get(cell, "column_id", "columnId", "column")
|
|
3606
|
+
)
|
|
3607
|
+
column = id_to_column.get(str(column_id)) or name_to_column.get(
|
|
3608
|
+
str(column_id)
|
|
3609
|
+
)
|
|
3610
|
+
if not column:
|
|
3611
|
+
continue
|
|
3612
|
+
column_name = str(column["name"])
|
|
3613
|
+
values[column_name] = _futureagi_parse_cell(
|
|
3614
|
+
_futureagi_get(cell, "value", "cell_value", "cellValue"),
|
|
3615
|
+
column=column,
|
|
3616
|
+
)
|
|
3617
|
+
return values, metadata
|
|
3618
|
+
|
|
3619
|
+
if isinstance(row, Mapping):
|
|
3620
|
+
for column in columns:
|
|
3621
|
+
column_id = str(column.get("id") or "")
|
|
3622
|
+
column_name = str(column.get("name") or "")
|
|
3623
|
+
raw_cell = None
|
|
3624
|
+
if column_id and column_id in row:
|
|
3625
|
+
raw_cell = row[column_id]
|
|
3626
|
+
elif column_name and column_name in row:
|
|
3627
|
+
raw_cell = row[column_name]
|
|
3628
|
+
else:
|
|
3629
|
+
continue
|
|
3630
|
+
values[column_name] = _futureagi_parse_cell(raw_cell, column=column)
|
|
3631
|
+
if not values:
|
|
3632
|
+
for column_name in (
|
|
3633
|
+
"case_id",
|
|
3634
|
+
"query",
|
|
3635
|
+
"response",
|
|
3636
|
+
"expected_response",
|
|
3637
|
+
"observability",
|
|
3638
|
+
"tags",
|
|
3639
|
+
"metadata",
|
|
3640
|
+
):
|
|
3641
|
+
if column_name in row:
|
|
3642
|
+
column = name_to_column.get(column_name) or {
|
|
3643
|
+
"name": column_name,
|
|
3644
|
+
"data_type": _futureagi_column_data_type(column_name),
|
|
3645
|
+
}
|
|
3646
|
+
values[column_name] = _futureagi_parse_cell(
|
|
3647
|
+
row[column_name],
|
|
3648
|
+
column=column,
|
|
3649
|
+
)
|
|
3650
|
+
return values, metadata
|
|
3651
|
+
|
|
3652
|
+
|
|
3653
|
+
def _futureagi_parse_cell(value: Any, *, column: Mapping[str, Any]) -> Any:
|
|
3654
|
+
value = _futureagi_cell_payload_value(value)
|
|
3655
|
+
column_name = str(column.get("name") or "")
|
|
3656
|
+
data_type = str(
|
|
3657
|
+
column.get("data_type")
|
|
3658
|
+
or _futureagi_column_data_type(column_name)
|
|
3659
|
+
or "text"
|
|
3660
|
+
).lower()
|
|
3661
|
+
if isinstance(value, str):
|
|
3662
|
+
stripped = value.strip()
|
|
3663
|
+
should_parse_json = (
|
|
3664
|
+
data_type in {"json", "array"}
|
|
3665
|
+
or column_name in {"expected_response", "observability", "tags", "metadata"}
|
|
3666
|
+
or stripped.startswith("{")
|
|
3667
|
+
or stripped.startswith("[")
|
|
3668
|
+
)
|
|
3669
|
+
if should_parse_json and stripped:
|
|
3670
|
+
try:
|
|
3671
|
+
return json.loads(stripped)
|
|
3672
|
+
except json.JSONDecodeError:
|
|
3673
|
+
return value
|
|
3674
|
+
return value
|
|
3675
|
+
|
|
3676
|
+
|
|
3677
|
+
def _futureagi_cell_payload_value(value: Any) -> Any:
|
|
3678
|
+
if hasattr(value, "model_dump"):
|
|
3679
|
+
value = value.model_dump()
|
|
3680
|
+
elif hasattr(value, "dict"):
|
|
3681
|
+
value = value.dict()
|
|
3682
|
+
if isinstance(value, Mapping):
|
|
3683
|
+
for key in ("cell_value", "cellValue", "value"):
|
|
3684
|
+
if key in value:
|
|
3685
|
+
return value[key]
|
|
3686
|
+
return value
|
|
3687
|
+
|
|
3688
|
+
|
|
3689
|
+
def _futureagi_column_data_type(column_name: str) -> str:
|
|
3690
|
+
for column in FUTUREAGI_REGRESSION_DATASET_COLUMNS:
|
|
3691
|
+
if column["name"] == column_name:
|
|
3692
|
+
return str(column["data_type"])
|
|
3693
|
+
return "text"
|
|
3694
|
+
|
|
3695
|
+
|
|
3696
|
+
def _futureagi_get(value: Any, *keys: str) -> Any:
|
|
3697
|
+
if hasattr(value, "model_dump"):
|
|
3698
|
+
value = value.model_dump()
|
|
3699
|
+
elif hasattr(value, "dict"):
|
|
3700
|
+
value = value.dict()
|
|
3701
|
+
if isinstance(value, Mapping):
|
|
3702
|
+
for key in keys:
|
|
3703
|
+
if key in value:
|
|
3704
|
+
return value[key]
|
|
3705
|
+
return None
|
|
3706
|
+
for key in keys:
|
|
3707
|
+
item = getattr(value, key, None)
|
|
3708
|
+
if item is not None:
|
|
3709
|
+
return item
|
|
3710
|
+
return None
|
|
3711
|
+
|
|
3712
|
+
|
|
3713
|
+
def _futureagi_scalar(value: Any) -> Optional[str]:
|
|
3714
|
+
if value is None:
|
|
3715
|
+
return None
|
|
3716
|
+
if hasattr(value, "value"):
|
|
3717
|
+
value = value.value
|
|
3718
|
+
return str(value)
|
|
3719
|
+
|
|
3720
|
+
|
|
3721
|
+
def _ensure_mapping(value: Any) -> dict[str, Any]:
|
|
3722
|
+
if hasattr(value, "model_dump"):
|
|
3723
|
+
value = value.model_dump()
|
|
3724
|
+
elif hasattr(value, "dict"):
|
|
3725
|
+
value = value.dict()
|
|
3726
|
+
if isinstance(value, Mapping):
|
|
3727
|
+
return dict(value)
|
|
3728
|
+
return {}
|
|
3729
|
+
|
|
3730
|
+
|
|
3731
|
+
def _string_list(value: Any) -> list[str]:
|
|
3732
|
+
if value is None:
|
|
3733
|
+
return []
|
|
3734
|
+
if isinstance(value, str):
|
|
3735
|
+
stripped = value.strip()
|
|
3736
|
+
if not stripped:
|
|
3737
|
+
return []
|
|
3738
|
+
if stripped.startswith("["):
|
|
3739
|
+
try:
|
|
3740
|
+
parsed = json.loads(stripped)
|
|
3741
|
+
except json.JSONDecodeError:
|
|
3742
|
+
parsed = None
|
|
3743
|
+
if isinstance(parsed, Sequence) and not isinstance(
|
|
3744
|
+
parsed,
|
|
3745
|
+
(str, bytes, bytearray),
|
|
3746
|
+
):
|
|
3747
|
+
return [str(item) for item in parsed if item is not None]
|
|
3748
|
+
return [stripped]
|
|
3749
|
+
if isinstance(value, Sequence) and not isinstance(value, (str, bytes, bytearray)):
|
|
3750
|
+
return [str(item) for item in value if item is not None]
|
|
3751
|
+
return [str(value)]
|
|
3752
|
+
|
|
3753
|
+
|
|
3754
|
+
def _float_mapping(value: Any) -> dict[str, float]:
|
|
3755
|
+
if not isinstance(value, Mapping):
|
|
3756
|
+
return {}
|
|
3757
|
+
metrics: dict[str, float] = {}
|
|
3758
|
+
for key, raw_value in value.items():
|
|
3759
|
+
score = _coerce_score(raw_value)
|
|
3760
|
+
if score is not None:
|
|
3761
|
+
metrics[str(key)] = score
|
|
3762
|
+
return metrics
|
|
3763
|
+
|
|
3764
|
+
|
|
3765
|
+
def _observability_record_from_regression_case(
|
|
3766
|
+
case: AgentRegressionCase,
|
|
3767
|
+
*,
|
|
3768
|
+
index: int,
|
|
3769
|
+
candidate: Optional[AgentCandidate],
|
|
3770
|
+
source: str,
|
|
3771
|
+
framework: str,
|
|
3772
|
+
) -> AgentObservabilityRecord:
|
|
3773
|
+
observability = _ensure_mapping(case.input.get("observability"))
|
|
3774
|
+
raw = observability.get("raw")
|
|
3775
|
+
if not isinstance(raw, Mapping):
|
|
3776
|
+
raw = copy.deepcopy(observability)
|
|
3777
|
+
failures = _string_list(
|
|
3778
|
+
observability.get("failures")
|
|
3779
|
+
or case.expected.get("previous_failures")
|
|
3780
|
+
or case.metadata.get("futureagi_response")
|
|
3781
|
+
)
|
|
3782
|
+
score = _coerce_score(observability.get("score"))
|
|
3783
|
+
if score is None:
|
|
3784
|
+
score = _coerce_score(case.expected.get("previous_score"))
|
|
3785
|
+
if score is None:
|
|
3786
|
+
score = 0.0 if failures else 1.0
|
|
3787
|
+
passed_value = observability.get("passed")
|
|
3788
|
+
if isinstance(passed_value, bool):
|
|
3789
|
+
passed = passed_value
|
|
3790
|
+
else:
|
|
3791
|
+
passed = not failures
|
|
3792
|
+
resolved_source = _normalize_source(
|
|
3793
|
+
observability.get("source")
|
|
3794
|
+
or case.metadata.get("source")
|
|
3795
|
+
or source
|
|
3796
|
+
or "futureagi"
|
|
3797
|
+
)
|
|
3798
|
+
resolved_framework = _normalize_source(
|
|
3799
|
+
observability.get("framework")
|
|
3800
|
+
or case.metadata.get("framework")
|
|
3801
|
+
or framework
|
|
3802
|
+
or "generic"
|
|
3803
|
+
)
|
|
3804
|
+
run_id = (
|
|
3805
|
+
observability.get("run_id")
|
|
3806
|
+
or case.metadata.get("run_id")
|
|
3807
|
+
or case.metadata.get("futureagi_row_id")
|
|
3808
|
+
or case.id
|
|
3809
|
+
)
|
|
3810
|
+
candidate_id = (
|
|
3811
|
+
observability.get("candidate_id")
|
|
3812
|
+
or case.metadata.get("candidate_id")
|
|
3813
|
+
or (candidate.id if candidate is not None else None)
|
|
3814
|
+
)
|
|
3815
|
+
return AgentObservabilityRecord(
|
|
3816
|
+
index=index,
|
|
3817
|
+
source=resolved_source,
|
|
3818
|
+
framework=resolved_framework,
|
|
3819
|
+
run_id=str(run_id) if run_id is not None else None,
|
|
3820
|
+
candidate_id=str(candidate_id) if candidate_id is not None else None,
|
|
3821
|
+
score=score,
|
|
3822
|
+
passed=passed,
|
|
3823
|
+
failures=failures,
|
|
3824
|
+
metrics=_float_mapping(observability.get("metrics")),
|
|
3825
|
+
trace_signals=_string_list(observability.get("trace_signals")),
|
|
3826
|
+
raw=copy.deepcopy(dict(raw)),
|
|
3827
|
+
metadata={
|
|
3828
|
+
"source_kind": resolved_source,
|
|
3829
|
+
"framework": resolved_framework,
|
|
3830
|
+
"regression_case_id": case.id,
|
|
3831
|
+
"regression_case_tags": list(case.tags),
|
|
3832
|
+
"regression_case_metadata": copy.deepcopy(case.metadata),
|
|
3833
|
+
},
|
|
3834
|
+
)
|
|
3835
|
+
|
|
3836
|
+
|
|
3837
|
+
def _agent_report_case_raw_evidence(case: Mapping[str, Any]) -> dict[str, Any]:
|
|
3838
|
+
observability = _ensure_mapping(case.get("observability"))
|
|
3839
|
+
if not observability:
|
|
3840
|
+
observability = _ensure_mapping(_ensure_mapping(case.get("input")).get("observability"))
|
|
3841
|
+
raw = _ensure_mapping(observability.get("raw"))
|
|
3842
|
+
if not raw:
|
|
3843
|
+
raw = _ensure_mapping(case.get("raw"))
|
|
3844
|
+
return copy.deepcopy(raw)
|
|
3845
|
+
|
|
3846
|
+
|
|
3847
|
+
def _agent_report_evaluation_metrics(evaluation: Any) -> dict[str, float]:
|
|
3848
|
+
payload = _ensure_mapping(evaluation)
|
|
3849
|
+
metrics = _float_mapping(_ensure_mapping(_ensure_mapping(payload.get("summary")).get("metric_averages")))
|
|
3850
|
+
if metrics:
|
|
3851
|
+
return metrics
|
|
3852
|
+
for case in _sequence_items(payload.get("cases")):
|
|
3853
|
+
case_dict = _ensure_mapping(case)
|
|
3854
|
+
for metric in _sequence_items(case_dict.get("metrics")):
|
|
3855
|
+
metric_dict = _ensure_mapping(metric)
|
|
3856
|
+
name = metric_dict.get("name")
|
|
3857
|
+
score = _coerce_score(metric_dict.get("score"))
|
|
3858
|
+
if name and score is not None:
|
|
3859
|
+
metrics[str(name)] = score
|
|
3860
|
+
return metrics
|
|
3861
|
+
|
|
3862
|
+
|
|
3863
|
+
def _agent_report_evaluation_failures(evaluation: Any) -> list[str]:
|
|
3864
|
+
payload = _ensure_mapping(evaluation)
|
|
3865
|
+
failures: list[str] = []
|
|
3866
|
+
for finding in _sequence_items(payload.get("findings")):
|
|
3867
|
+
finding_dict = _ensure_mapping(finding)
|
|
3868
|
+
finding_type = finding_dict.get("type") or finding_dict.get("metric") or finding_dict.get("reason")
|
|
3869
|
+
if finding_type:
|
|
3870
|
+
failures.append(str(finding_type))
|
|
3871
|
+
for case in _sequence_items(payload.get("cases")):
|
|
3872
|
+
case_dict = _ensure_mapping(case)
|
|
3873
|
+
for finding in _sequence_items(case_dict.get("findings")):
|
|
3874
|
+
finding_dict = _ensure_mapping(finding)
|
|
3875
|
+
finding_type = finding_dict.get("type") or finding_dict.get("metric") or finding_dict.get("reason")
|
|
3876
|
+
if finding_type:
|
|
3877
|
+
failures.append(str(finding_type))
|
|
3878
|
+
for metric in _sequence_items(case_dict.get("metrics")):
|
|
3879
|
+
metric_dict = _ensure_mapping(metric)
|
|
3880
|
+
score = _coerce_score(metric_dict.get("score"))
|
|
3881
|
+
if score is not None and score < 1.0 and metric_dict.get("reason"):
|
|
3882
|
+
failures.append(str(metric_dict["reason"]))
|
|
3883
|
+
return list(dict.fromkeys(failures))
|
|
3884
|
+
|
|
3885
|
+
|
|
3886
|
+
def _metric_threshold_failures(
|
|
3887
|
+
metrics: Mapping[str, float],
|
|
3888
|
+
thresholds: Mapping[str, float],
|
|
3889
|
+
) -> list[str]:
|
|
3890
|
+
failures: list[str] = []
|
|
3891
|
+
for name, threshold in thresholds.items():
|
|
3892
|
+
observed = metrics.get(name)
|
|
3893
|
+
if observed is None:
|
|
3894
|
+
failures.append(f"metric '{name}' missing from agent-report replay case")
|
|
3895
|
+
elif observed < threshold:
|
|
3896
|
+
failures.append(
|
|
3897
|
+
f"metric '{name}' below required threshold {threshold:.4f}: {observed:.4f}"
|
|
3898
|
+
)
|
|
3899
|
+
return failures
|
|
3900
|
+
|
|
3901
|
+
|
|
3902
|
+
def _sequence_items(value: Any) -> list[Any]:
|
|
3903
|
+
if value is None:
|
|
3904
|
+
return []
|
|
3905
|
+
if isinstance(value, Sequence) and not isinstance(value, (str, bytes, bytearray)):
|
|
3906
|
+
return list(value)
|
|
3907
|
+
return [value]
|
|
3908
|
+
|
|
3909
|
+
|
|
3910
|
+
def _regression_dataset_required_metrics(
|
|
3911
|
+
cases: Sequence[AgentRegressionCase],
|
|
3912
|
+
*,
|
|
3913
|
+
override: Optional[Mapping[str, float]],
|
|
3914
|
+
) -> dict[str, float]:
|
|
3915
|
+
if override is not None:
|
|
3916
|
+
return {str(key): float(value) for key, value in dict(override).items()}
|
|
3917
|
+
thresholds: dict[str, float] = {}
|
|
3918
|
+
for case in cases:
|
|
3919
|
+
expected = _ensure_mapping(case.expected)
|
|
3920
|
+
for key, value in _ensure_mapping(expected.get("required_metrics")).items():
|
|
3921
|
+
try:
|
|
3922
|
+
thresholds.setdefault(str(key), float(value))
|
|
3923
|
+
except (TypeError, ValueError):
|
|
3924
|
+
continue
|
|
3925
|
+
return thresholds
|
|
3926
|
+
|
|
3927
|
+
|
|
3928
|
+
def _regression_dataset_required_trace_signals(
|
|
3929
|
+
cases: Sequence[AgentRegressionCase],
|
|
3930
|
+
*,
|
|
3931
|
+
override: Optional[Sequence[str]],
|
|
3932
|
+
) -> list[str]:
|
|
3933
|
+
if override is not None:
|
|
3934
|
+
return [_normalize_signal(item) for item in override if _normalize_signal(item)]
|
|
3935
|
+
signals: list[str] = []
|
|
3936
|
+
seen: set[str] = set()
|
|
3937
|
+
for case in cases:
|
|
3938
|
+
expected = _ensure_mapping(case.expected)
|
|
3939
|
+
for signal in _string_list(expected.get("required_trace_signals")):
|
|
3940
|
+
normalized = _normalize_signal(signal)
|
|
3941
|
+
if normalized and normalized not in seen:
|
|
3942
|
+
signals.append(normalized)
|
|
3943
|
+
seen.add(normalized)
|
|
3944
|
+
return signals
|
|
3945
|
+
|
|
3946
|
+
|
|
3947
|
+
def _regression_cases_source(cases: Sequence[AgentRegressionCase]) -> str:
|
|
3948
|
+
sources = sorted(
|
|
3949
|
+
{
|
|
3950
|
+
_normalize_source(
|
|
3951
|
+
_ensure_mapping(case.input.get("observability")).get("source")
|
|
3952
|
+
or case.metadata.get("source")
|
|
3953
|
+
or "futureagi"
|
|
3954
|
+
)
|
|
3955
|
+
for case in cases
|
|
3956
|
+
}
|
|
3957
|
+
)
|
|
3958
|
+
if not sources:
|
|
3959
|
+
return "futureagi"
|
|
3960
|
+
return sources[0] if len(sources) == 1 else "mixed"
|
|
3961
|
+
|
|
3962
|
+
|
|
3963
|
+
def _regression_cases_framework(cases: Sequence[AgentRegressionCase]) -> str:
|
|
3964
|
+
frameworks = sorted(
|
|
3965
|
+
{
|
|
3966
|
+
_normalize_source(
|
|
3967
|
+
_ensure_mapping(case.input.get("observability")).get("framework")
|
|
3968
|
+
or case.metadata.get("framework")
|
|
3969
|
+
or "generic"
|
|
3970
|
+
)
|
|
3971
|
+
for case in cases
|
|
3972
|
+
}
|
|
3973
|
+
)
|
|
3974
|
+
if not frameworks:
|
|
3975
|
+
return "generic"
|
|
3976
|
+
return frameworks[0] if len(frameworks) == 1 else "mixed"
|
|
3977
|
+
|
|
3978
|
+
|
|
3979
|
+
def _normalize_observability_record(
|
|
3980
|
+
record: dict[str, Any],
|
|
3981
|
+
*,
|
|
3982
|
+
index: int,
|
|
3983
|
+
candidate: Optional[AgentCandidate],
|
|
3984
|
+
source: str,
|
|
3985
|
+
framework: str,
|
|
3986
|
+
required_metrics: Mapping[str, float],
|
|
3987
|
+
required_trace_signals: Sequence[str],
|
|
3988
|
+
) -> AgentObservabilityRecord:
|
|
3989
|
+
resolved_source = _resolve_source(record, fallback=source)
|
|
3990
|
+
resolved_framework = _resolve_framework(record, fallback=framework)
|
|
3991
|
+
trace_items = _trace_items(record)
|
|
3992
|
+
trace_signals = sorted(_trace_signals(record, trace_items))
|
|
3993
|
+
metrics = _extract_metrics(record)
|
|
3994
|
+
if trace_items:
|
|
3995
|
+
metrics.setdefault(
|
|
3996
|
+
"framework_trace_coverage",
|
|
3997
|
+
_trace_coverage(trace_signals, required_trace_signals),
|
|
3998
|
+
)
|
|
3999
|
+
if _has_transcript(record):
|
|
4000
|
+
metrics.setdefault(
|
|
4001
|
+
"framework_transcript_quality",
|
|
4002
|
+
0.0 if _has_error(record, trace_items) else 1.0,
|
|
4003
|
+
)
|
|
4004
|
+
if _has_error(record, trace_items):
|
|
4005
|
+
metrics.setdefault("runtime_success", 0.0)
|
|
4006
|
+
|
|
4007
|
+
failures = _record_failures(
|
|
4008
|
+
record,
|
|
4009
|
+
metrics=metrics,
|
|
4010
|
+
trace_signals=trace_signals,
|
|
4011
|
+
required_metrics=required_metrics,
|
|
4012
|
+
required_trace_signals=required_trace_signals,
|
|
4013
|
+
)
|
|
4014
|
+
score = _record_score(record, metrics=metrics, failures=failures)
|
|
4015
|
+
passed = not failures
|
|
4016
|
+
return AgentObservabilityRecord(
|
|
4017
|
+
index=index,
|
|
4018
|
+
source=resolved_source,
|
|
4019
|
+
framework=resolved_framework,
|
|
4020
|
+
run_id=_first_string(record, "run_id", "id", "trace_id", "session_id", "room_name"),
|
|
4021
|
+
candidate_id=_candidate_id(record, candidate),
|
|
4022
|
+
score=score,
|
|
4023
|
+
passed=passed,
|
|
4024
|
+
failures=failures,
|
|
4025
|
+
metrics=metrics,
|
|
4026
|
+
trace_signals=trace_signals,
|
|
4027
|
+
raw=copy.deepcopy(record),
|
|
4028
|
+
metadata={
|
|
4029
|
+
"source_kind": resolved_source,
|
|
4030
|
+
"framework": resolved_framework,
|
|
4031
|
+
"trace_item_count": len(trace_items),
|
|
4032
|
+
},
|
|
4033
|
+
)
|
|
4034
|
+
|
|
4035
|
+
|
|
4036
|
+
def _evaluation_from_observability_record(
|
|
4037
|
+
record: AgentObservabilityRecord,
|
|
4038
|
+
*,
|
|
4039
|
+
candidate: Optional[AgentCandidate],
|
|
4040
|
+
) -> CandidateEvaluation:
|
|
4041
|
+
evaluation_candidate = candidate or AgentCandidate.from_config(
|
|
4042
|
+
record.raw.get("candidate_config")
|
|
4043
|
+
or record.raw.get("config")
|
|
4044
|
+
or {"observability": {"source": record.source, "framework": record.framework}},
|
|
4045
|
+
target_name=str(record.raw.get("target_name") or "observability-feedback"),
|
|
4046
|
+
metadata={
|
|
4047
|
+
"kind": "observability_feedback",
|
|
4048
|
+
"observability_source": record.source,
|
|
4049
|
+
"observability_framework": record.framework,
|
|
4050
|
+
"observability_run_id": record.run_id,
|
|
4051
|
+
},
|
|
4052
|
+
)
|
|
4053
|
+
return CandidateEvaluation(
|
|
4054
|
+
candidate=evaluation_candidate,
|
|
4055
|
+
score=record.score,
|
|
4056
|
+
reason="; ".join(record.failures),
|
|
4057
|
+
metadata={
|
|
4058
|
+
"agent_observability_feedback": record.model_dump(),
|
|
4059
|
+
"agent_report_evaluation": _agent_report_from_observability_record(record),
|
|
4060
|
+
},
|
|
4061
|
+
)
|
|
4062
|
+
|
|
4063
|
+
|
|
4064
|
+
def _agent_report_from_observability_record(record: AgentObservabilityRecord) -> dict[str, Any]:
|
|
4065
|
+
return {
|
|
4066
|
+
"summary": {"metric_averages": dict(record.metrics)},
|
|
4067
|
+
"cases": [
|
|
4068
|
+
{
|
|
4069
|
+
"id": record.run_id or f"observability-{record.index}",
|
|
4070
|
+
"metrics": [
|
|
4071
|
+
{
|
|
4072
|
+
"name": name,
|
|
4073
|
+
"score": score,
|
|
4074
|
+
"reason": "; ".join(record.failures),
|
|
4075
|
+
}
|
|
4076
|
+
for name, score in record.metrics.items()
|
|
4077
|
+
],
|
|
4078
|
+
"findings": [
|
|
4079
|
+
{
|
|
4080
|
+
"metric": name,
|
|
4081
|
+
"score": score,
|
|
4082
|
+
"evidence": "; ".join(record.failures),
|
|
4083
|
+
}
|
|
4084
|
+
for name, score in record.metrics.items()
|
|
4085
|
+
],
|
|
4086
|
+
}
|
|
4087
|
+
],
|
|
4088
|
+
}
|
|
4089
|
+
|
|
4090
|
+
|
|
4091
|
+
def _regression_windows(
|
|
4092
|
+
windows: AgentObservabilityWindow | Sequence[AgentObservabilityWindow],
|
|
4093
|
+
) -> list[AgentObservabilityWindow]:
|
|
4094
|
+
if isinstance(windows, AgentObservabilityWindow):
|
|
4095
|
+
return [windows]
|
|
4096
|
+
if isinstance(windows, Sequence) and not isinstance(windows, (str, bytes, bytearray)):
|
|
4097
|
+
return list(windows)
|
|
4098
|
+
raise TypeError("windows must be an AgentObservabilityWindow or a sequence of windows")
|
|
4099
|
+
|
|
4100
|
+
|
|
4101
|
+
def _regression_case_from_observability_record(
|
|
4102
|
+
record: AgentObservabilityRecord,
|
|
4103
|
+
*,
|
|
4104
|
+
window: AgentObservabilityWindow,
|
|
4105
|
+
window_index: int,
|
|
4106
|
+
include_raw: bool,
|
|
4107
|
+
) -> AgentRegressionCase:
|
|
4108
|
+
case_id = "-".join(
|
|
4109
|
+
item
|
|
4110
|
+
for item in (
|
|
4111
|
+
_case_slug(record.source),
|
|
4112
|
+
_case_slug(record.framework),
|
|
4113
|
+
_case_slug(record.run_id or f"record-{record.index}"),
|
|
4114
|
+
str(record.index),
|
|
4115
|
+
)
|
|
4116
|
+
if item
|
|
4117
|
+
)
|
|
4118
|
+
observability_input = {
|
|
4119
|
+
"source": record.source,
|
|
4120
|
+
"framework": record.framework,
|
|
4121
|
+
"run_id": record.run_id,
|
|
4122
|
+
"candidate_id": record.candidate_id,
|
|
4123
|
+
"score": record.score,
|
|
4124
|
+
"passed": record.passed,
|
|
4125
|
+
"failures": list(record.failures),
|
|
4126
|
+
"metrics": dict(record.metrics),
|
|
4127
|
+
"trace_signals": list(record.trace_signals),
|
|
4128
|
+
}
|
|
4129
|
+
if include_raw:
|
|
4130
|
+
observability_input["raw"] = copy.deepcopy(record.raw)
|
|
4131
|
+
|
|
4132
|
+
expected = {
|
|
4133
|
+
"should_pass": True,
|
|
4134
|
+
"required_metrics": dict(window.required_metrics),
|
|
4135
|
+
"required_trace_signals": list(window.required_trace_signals),
|
|
4136
|
+
"previous_score": record.score,
|
|
4137
|
+
"previous_failures": list(record.failures),
|
|
4138
|
+
}
|
|
4139
|
+
return AgentRegressionCase(
|
|
4140
|
+
id=case_id,
|
|
4141
|
+
input={"observability": observability_input},
|
|
4142
|
+
expected=expected,
|
|
4143
|
+
tags=_regression_tags(record, window=window),
|
|
4144
|
+
metadata={
|
|
4145
|
+
"kind": "observability_regression_case",
|
|
4146
|
+
"source": record.source,
|
|
4147
|
+
"framework": record.framework,
|
|
4148
|
+
"window_source": window.source,
|
|
4149
|
+
"window_framework": window.framework,
|
|
4150
|
+
"window_index": window_index,
|
|
4151
|
+
"record_index": record.index,
|
|
4152
|
+
"run_id": record.run_id,
|
|
4153
|
+
"candidate_id": record.candidate_id,
|
|
4154
|
+
"observed_score": record.score,
|
|
4155
|
+
"passed": record.passed,
|
|
4156
|
+
"failure_count": len(record.failures),
|
|
4157
|
+
"record_metadata": copy.deepcopy(record.metadata),
|
|
4158
|
+
"window_metadata": copy.deepcopy(window.metadata),
|
|
4159
|
+
},
|
|
4160
|
+
)
|
|
4161
|
+
|
|
4162
|
+
|
|
4163
|
+
def _regression_tags(
|
|
4164
|
+
record: AgentObservabilityRecord,
|
|
4165
|
+
*,
|
|
4166
|
+
window: AgentObservabilityWindow,
|
|
4167
|
+
) -> list[str]:
|
|
4168
|
+
tags = {
|
|
4169
|
+
"observability",
|
|
4170
|
+
f"source:{record.source}",
|
|
4171
|
+
f"framework:{record.framework}",
|
|
4172
|
+
"status:passed" if record.passed else "status:failed",
|
|
4173
|
+
}
|
|
4174
|
+
for metric, threshold in window.required_metrics.items():
|
|
4175
|
+
observed = record.metrics.get(metric)
|
|
4176
|
+
if observed is None or observed < threshold:
|
|
4177
|
+
tags.add(f"metric:{_case_slug(metric)}")
|
|
4178
|
+
present_signals = set(record.trace_signals)
|
|
4179
|
+
for signal in window.required_trace_signals:
|
|
4180
|
+
if signal not in present_signals:
|
|
4181
|
+
tags.add(f"missing_signal:{_case_slug(signal)}")
|
|
4182
|
+
if any("runtime error" in failure for failure in record.failures):
|
|
4183
|
+
tags.add("runtime:error")
|
|
4184
|
+
return sorted(tags)
|
|
4185
|
+
|
|
4186
|
+
|
|
4187
|
+
def _regression_source(windows: Sequence[AgentObservabilityWindow]) -> str:
|
|
4188
|
+
sources = sorted({window.source for window in windows})
|
|
4189
|
+
return sources[0] if len(sources) == 1 else "mixed"
|
|
4190
|
+
|
|
4191
|
+
|
|
4192
|
+
def _regression_framework(windows: Sequence[AgentObservabilityWindow]) -> str:
|
|
4193
|
+
frameworks = sorted({window.framework for window in windows})
|
|
4194
|
+
return frameworks[0] if len(frameworks) == 1 else "mixed"
|
|
4195
|
+
|
|
4196
|
+
|
|
4197
|
+
def _case_slug(value: Any) -> str:
|
|
4198
|
+
text = str(value or "").strip().lower()
|
|
4199
|
+
chars = [char if char.isalnum() else "-" for char in text]
|
|
4200
|
+
return "-".join(part for part in "".join(chars).split("-") if part)[:80]
|
|
4201
|
+
|
|
4202
|
+
|
|
4203
|
+
def _load_payload(payload: Any) -> Any:
|
|
4204
|
+
if hasattr(payload, "model_dump"):
|
|
4205
|
+
return payload.model_dump()
|
|
4206
|
+
if hasattr(payload, "dict"):
|
|
4207
|
+
return payload.dict()
|
|
4208
|
+
if isinstance(payload, Path):
|
|
4209
|
+
return _parse_observability_text(payload.read_text())
|
|
4210
|
+
if isinstance(payload, str):
|
|
4211
|
+
if _looks_like_observability_text(payload):
|
|
4212
|
+
return _parse_observability_text(payload)
|
|
4213
|
+
try:
|
|
4214
|
+
path = Path(payload)
|
|
4215
|
+
if path.exists() and path.is_file():
|
|
4216
|
+
return _parse_observability_text(path.read_text())
|
|
4217
|
+
except OSError:
|
|
4218
|
+
pass
|
|
4219
|
+
return _parse_observability_text(payload)
|
|
4220
|
+
return payload
|
|
4221
|
+
|
|
4222
|
+
|
|
4223
|
+
def _parse_observability_text(text: str) -> Any:
|
|
4224
|
+
stripped = text.strip()
|
|
4225
|
+
if not stripped:
|
|
4226
|
+
return []
|
|
4227
|
+
try:
|
|
4228
|
+
return json.loads(stripped)
|
|
4229
|
+
except json.JSONDecodeError:
|
|
4230
|
+
records = []
|
|
4231
|
+
for line in stripped.splitlines():
|
|
4232
|
+
line = line.strip()
|
|
4233
|
+
if not line:
|
|
4234
|
+
continue
|
|
4235
|
+
records.append(json.loads(line))
|
|
4236
|
+
return records
|
|
4237
|
+
|
|
4238
|
+
|
|
4239
|
+
def _looks_like_observability_text(text: str) -> bool:
|
|
4240
|
+
stripped = text.strip()
|
|
4241
|
+
return (
|
|
4242
|
+
stripped.startswith("{")
|
|
4243
|
+
or stripped.startswith("[")
|
|
4244
|
+
or "\n" in stripped
|
|
4245
|
+
or "\r" in stripped
|
|
4246
|
+
)
|
|
4247
|
+
|
|
4248
|
+
|
|
4249
|
+
def _observation_records(payload: Any) -> list[Any]:
|
|
4250
|
+
if hasattr(payload, "model_dump"):
|
|
4251
|
+
payload = payload.model_dump()
|
|
4252
|
+
if isinstance(payload, Sequence) and not isinstance(payload, (str, bytes, bytearray)):
|
|
4253
|
+
return list(payload)
|
|
4254
|
+
if not isinstance(payload, Mapping):
|
|
4255
|
+
return [payload]
|
|
4256
|
+
if "resourceSpans" in payload or "resource_spans" in payload:
|
|
4257
|
+
return [payload]
|
|
4258
|
+
for key in ("runs", "traces", "sessions", "records", "items", "results"):
|
|
4259
|
+
value = payload.get(key)
|
|
4260
|
+
if isinstance(value, Sequence) and not isinstance(value, (str, bytes, bytearray)):
|
|
4261
|
+
return list(value)
|
|
4262
|
+
return [payload]
|
|
4263
|
+
|
|
4264
|
+
|
|
4265
|
+
def _resolve_source(record: Mapping[str, Any], *, fallback: str) -> str:
|
|
4266
|
+
if fallback and fallback != "auto":
|
|
4267
|
+
return _normalize_source(fallback)
|
|
4268
|
+
explicit = _first_string(record, "source", "provider", "observability_source")
|
|
4269
|
+
if explicit:
|
|
4270
|
+
return _normalize_source(explicit)
|
|
4271
|
+
keys = {str(key).lower() for key in record}
|
|
4272
|
+
if {"feedback", "run_type", "dotted_order", "parent_run_id"} & keys:
|
|
4273
|
+
return "generic"
|
|
4274
|
+
if {"resourceSpans", "resource_spans"} & set(record):
|
|
4275
|
+
return "opentelemetry"
|
|
4276
|
+
if "span_data" in _json_text(record) or "trace_id" in keys:
|
|
4277
|
+
return "openai_agents"
|
|
4278
|
+
if {"room", "job", "session", "participant"} & keys or "makeSessionReport" in _json_text(record):
|
|
4279
|
+
return "livekit"
|
|
4280
|
+
return "generic"
|
|
4281
|
+
|
|
4282
|
+
|
|
4283
|
+
def _resolve_framework(record: Mapping[str, Any], *, fallback: str) -> str:
|
|
4284
|
+
if fallback and fallback != "auto":
|
|
4285
|
+
return _normalize_source(fallback)
|
|
4286
|
+
explicit = _first_string(record, "framework", "runtime", "sdk", "provider")
|
|
4287
|
+
if explicit:
|
|
4288
|
+
return _normalize_source(explicit)
|
|
4289
|
+
text = _json_text(record)
|
|
4290
|
+
checks = (
|
|
4291
|
+
("langgraph", ("langgraph", "stream_events")),
|
|
4292
|
+
("langchain", ("langchain", "stream_events")),
|
|
4293
|
+
("openai_agents", ("openai agents", "openai_agents", "span_data")),
|
|
4294
|
+
("livekit", ("livekit", "agent_session", "room")),
|
|
4295
|
+
("pipecat", ("pipecat", "frame")),
|
|
4296
|
+
("crewai", ("crewai", "crew")),
|
|
4297
|
+
("autogen", ("autogen", "groupchat")),
|
|
4298
|
+
("opentelemetry", ("resourceSpans", "gen_ai.")),
|
|
4299
|
+
)
|
|
4300
|
+
for name, tokens in checks:
|
|
4301
|
+
if any(token.lower() in text for token in tokens):
|
|
4302
|
+
return name
|
|
4303
|
+
return "generic"
|
|
4304
|
+
|
|
4305
|
+
|
|
4306
|
+
def _resolve_window_source(
|
|
4307
|
+
records: Sequence[AgentObservabilityRecord],
|
|
4308
|
+
*,
|
|
4309
|
+
fallback: str,
|
|
4310
|
+
) -> str:
|
|
4311
|
+
if fallback and fallback != "auto":
|
|
4312
|
+
return _normalize_source(fallback)
|
|
4313
|
+
sources = sorted({record.source for record in records})
|
|
4314
|
+
return sources[0] if len(sources) == 1 else "mixed"
|
|
4315
|
+
|
|
4316
|
+
|
|
4317
|
+
def _resolve_window_framework(
|
|
4318
|
+
records: Sequence[AgentObservabilityRecord],
|
|
4319
|
+
*,
|
|
4320
|
+
fallback: str,
|
|
4321
|
+
) -> str:
|
|
4322
|
+
if fallback and fallback != "auto":
|
|
4323
|
+
return _normalize_source(fallback)
|
|
4324
|
+
frameworks = sorted({record.framework for record in records})
|
|
4325
|
+
return frameworks[0] if len(frameworks) == 1 else "mixed"
|
|
4326
|
+
|
|
4327
|
+
|
|
4328
|
+
def _extract_metrics(record: Mapping[str, Any]) -> dict[str, float]:
|
|
4329
|
+
metrics: dict[str, float] = {}
|
|
4330
|
+
for path in (
|
|
4331
|
+
("metrics",),
|
|
4332
|
+
("metric_averages",),
|
|
4333
|
+
("scores",),
|
|
4334
|
+
("outputs", "metrics"),
|
|
4335
|
+
("outputs", "scores"),
|
|
4336
|
+
("metadata", "metrics"),
|
|
4337
|
+
("agent_report_evaluation", "summary", "metric_averages"),
|
|
4338
|
+
("evaluation", "summary", "metric_averages"),
|
|
4339
|
+
):
|
|
4340
|
+
value = _nested_get(record, path)
|
|
4341
|
+
if isinstance(value, Mapping):
|
|
4342
|
+
_merge_metric_mapping(metrics, value)
|
|
4343
|
+
|
|
4344
|
+
feedback = record.get("feedback")
|
|
4345
|
+
if isinstance(feedback, Mapping):
|
|
4346
|
+
_merge_metric_mapping(metrics, feedback)
|
|
4347
|
+
elif isinstance(feedback, Sequence) and not isinstance(feedback, (str, bytes, bytearray)):
|
|
4348
|
+
for item in feedback:
|
|
4349
|
+
_merge_metric_item(metrics, item)
|
|
4350
|
+
|
|
4351
|
+
for key in ("evaluations", "evaluation_results", "scores"):
|
|
4352
|
+
value = record.get(key)
|
|
4353
|
+
if isinstance(value, Sequence) and not isinstance(value, (str, bytes, bytearray)):
|
|
4354
|
+
for item in value:
|
|
4355
|
+
_merge_metric_item(metrics, item)
|
|
4356
|
+
|
|
4357
|
+
for report_key in ("agent_report_evaluation", "evaluation"):
|
|
4358
|
+
value = record.get(report_key)
|
|
4359
|
+
if isinstance(value, Mapping):
|
|
4360
|
+
_merge_agent_report_case_metrics(metrics, value)
|
|
4361
|
+
|
|
4362
|
+
explicit_score = _coerce_score(record.get("score"))
|
|
4363
|
+
if explicit_score is not None and not metrics:
|
|
4364
|
+
metrics["score"] = explicit_score
|
|
4365
|
+
return metrics
|
|
4366
|
+
|
|
4367
|
+
|
|
4368
|
+
def _merge_metric_mapping(metrics: dict[str, float], value: Mapping[str, Any]) -> None:
|
|
4369
|
+
for key, raw_score in value.items():
|
|
4370
|
+
score = _metric_score(raw_score)
|
|
4371
|
+
if score is not None:
|
|
4372
|
+
metrics[str(key)] = score
|
|
4373
|
+
|
|
4374
|
+
|
|
4375
|
+
def _merge_metric_item(metrics: dict[str, float], item: Any) -> None:
|
|
4376
|
+
if not isinstance(item, Mapping):
|
|
4377
|
+
return
|
|
4378
|
+
name = item.get("key") or item.get("name") or item.get("metric")
|
|
4379
|
+
score = _metric_score(
|
|
4380
|
+
item.get("score", item.get("value", item.get("output")))
|
|
4381
|
+
)
|
|
4382
|
+
if name and score is not None:
|
|
4383
|
+
metrics[str(name)] = score
|
|
4384
|
+
|
|
4385
|
+
|
|
4386
|
+
def _merge_agent_report_case_metrics(metrics: dict[str, float], report: Mapping[str, Any]) -> None:
|
|
4387
|
+
for case in report.get("cases", []) or []:
|
|
4388
|
+
if not isinstance(case, Mapping):
|
|
4389
|
+
continue
|
|
4390
|
+
for item in case.get("metrics", []) or []:
|
|
4391
|
+
_merge_metric_item(metrics, item)
|
|
4392
|
+
|
|
4393
|
+
|
|
4394
|
+
def _metric_score(value: Any) -> Optional[float]:
|
|
4395
|
+
if isinstance(value, Mapping):
|
|
4396
|
+
for key in ("score", "value", "output"):
|
|
4397
|
+
score = _coerce_score(value.get(key))
|
|
4398
|
+
if score is not None:
|
|
4399
|
+
return score
|
|
4400
|
+
return None
|
|
4401
|
+
return _coerce_score(value)
|
|
4402
|
+
|
|
4403
|
+
|
|
4404
|
+
def _coerce_score(value: Any) -> Optional[float]:
|
|
4405
|
+
if isinstance(value, bool):
|
|
4406
|
+
return 1.0 if value else 0.0
|
|
4407
|
+
if isinstance(value, (int, float)):
|
|
4408
|
+
return max(0.0, min(float(value), 1.0))
|
|
4409
|
+
return None
|
|
4410
|
+
|
|
4411
|
+
|
|
4412
|
+
def _record_score(
|
|
4413
|
+
record: Mapping[str, Any],
|
|
4414
|
+
*,
|
|
4415
|
+
metrics: Mapping[str, float],
|
|
4416
|
+
failures: Sequence[str],
|
|
4417
|
+
) -> float:
|
|
4418
|
+
explicit = _coerce_score(record.get("score"))
|
|
4419
|
+
if explicit is not None:
|
|
4420
|
+
return explicit
|
|
4421
|
+
if metrics:
|
|
4422
|
+
return sum(metrics.values()) / len(metrics)
|
|
4423
|
+
return 0.0 if failures else 1.0
|
|
4424
|
+
|
|
4425
|
+
|
|
4426
|
+
def _record_failures(
|
|
4427
|
+
record: Mapping[str, Any],
|
|
4428
|
+
*,
|
|
4429
|
+
metrics: Mapping[str, float],
|
|
4430
|
+
trace_signals: Sequence[str],
|
|
4431
|
+
required_metrics: Mapping[str, float],
|
|
4432
|
+
required_trace_signals: Sequence[str],
|
|
4433
|
+
) -> list[str]:
|
|
4434
|
+
failures: list[str] = []
|
|
4435
|
+
for name, threshold in required_metrics.items():
|
|
4436
|
+
observed = metrics.get(name)
|
|
4437
|
+
if observed is None:
|
|
4438
|
+
failures.append(f"metric '{name}' missing from observability record")
|
|
4439
|
+
elif observed < threshold:
|
|
4440
|
+
failures.append(
|
|
4441
|
+
f"metric '{name}' score {observed:.4f} below {threshold:.4f}"
|
|
4442
|
+
)
|
|
4443
|
+
missing_signals = [
|
|
4444
|
+
signal for signal in required_trace_signals if signal not in trace_signals
|
|
4445
|
+
]
|
|
4446
|
+
if missing_signals:
|
|
4447
|
+
failures.append(
|
|
4448
|
+
"missing trace signal(s): " + ", ".join(sorted(missing_signals))
|
|
4449
|
+
)
|
|
4450
|
+
if _has_error(record, _trace_items(record)):
|
|
4451
|
+
failures.append("observability record contains runtime error signal")
|
|
4452
|
+
return failures
|
|
4453
|
+
|
|
4454
|
+
|
|
4455
|
+
def _trace_items(record: Mapping[str, Any]) -> list[Any]:
|
|
4456
|
+
items: list[Any] = []
|
|
4457
|
+
for key in ("spans", "events", "session_events", "trace_events"):
|
|
4458
|
+
value = record.get(key)
|
|
4459
|
+
if isinstance(value, Sequence) and not isinstance(value, (str, bytes, bytearray)):
|
|
4460
|
+
items.extend(value)
|
|
4461
|
+
trace = record.get("trace")
|
|
4462
|
+
if isinstance(trace, Mapping):
|
|
4463
|
+
for key in ("spans", "events"):
|
|
4464
|
+
value = trace.get(key)
|
|
4465
|
+
if isinstance(value, Sequence) and not isinstance(value, (str, bytes, bytearray)):
|
|
4466
|
+
items.extend(value)
|
|
4467
|
+
report = record.get("report") or record.get("session_report") or record.get("session")
|
|
4468
|
+
if isinstance(report, Mapping):
|
|
4469
|
+
for key in ("events", "history", "conversation"):
|
|
4470
|
+
value = report.get(key)
|
|
4471
|
+
if isinstance(value, Sequence) and not isinstance(value, (str, bytes, bytearray)):
|
|
4472
|
+
items.extend(value)
|
|
4473
|
+
items.extend(_otlp_spans(record))
|
|
4474
|
+
return items
|
|
4475
|
+
|
|
4476
|
+
|
|
4477
|
+
def _otlp_spans(record: Mapping[str, Any]) -> list[Any]:
|
|
4478
|
+
resource_spans = record.get("resourceSpans") or record.get("resource_spans") or []
|
|
4479
|
+
spans: list[Any] = []
|
|
4480
|
+
for resource in resource_spans:
|
|
4481
|
+
if not isinstance(resource, Mapping):
|
|
4482
|
+
continue
|
|
4483
|
+
scope_spans = resource.get("scopeSpans") or resource.get("scope_spans") or []
|
|
4484
|
+
for scope in scope_spans:
|
|
4485
|
+
if not isinstance(scope, Mapping):
|
|
4486
|
+
continue
|
|
4487
|
+
scope_items = scope.get("spans") or []
|
|
4488
|
+
if isinstance(scope_items, Sequence) and not isinstance(scope_items, (str, bytes, bytearray)):
|
|
4489
|
+
spans.extend(scope_items)
|
|
4490
|
+
return spans
|
|
4491
|
+
|
|
4492
|
+
|
|
4493
|
+
def _trace_signals(record: Mapping[str, Any], items: Sequence[Any]) -> set[str]:
|
|
4494
|
+
signals: set[str] = set()
|
|
4495
|
+
signal_items = list(items) or [record]
|
|
4496
|
+
for item in signal_items:
|
|
4497
|
+
text = _json_text(item)
|
|
4498
|
+
attributes = _attributes_text(item)
|
|
4499
|
+
combined = f"{text} {attributes}"
|
|
4500
|
+
if "invoke_agent" in combined or "agent" in combined:
|
|
4501
|
+
signals.add("agent")
|
|
4502
|
+
if "chat" in combined or "llm" in combined or "model" in combined:
|
|
4503
|
+
signals.add("model")
|
|
4504
|
+
if "execute_tool" in combined or "tool" in combined or "function_call" in combined:
|
|
4505
|
+
signals.add("tool")
|
|
4506
|
+
if "handoff" in combined or "delegate" in combined:
|
|
4507
|
+
signals.add("handoff")
|
|
4508
|
+
if "guardrail" in combined or "safety" in combined:
|
|
4509
|
+
signals.add("guardrail")
|
|
4510
|
+
if "message" in combined or "transcript" in combined or "conversation" in combined:
|
|
4511
|
+
signals.add("message")
|
|
4512
|
+
if "error" in combined or "exception" in combined or "failed" in combined:
|
|
4513
|
+
signals.add("error")
|
|
4514
|
+
return signals
|
|
4515
|
+
|
|
4516
|
+
|
|
4517
|
+
def _attributes_text(item: Any) -> str:
|
|
4518
|
+
if not isinstance(item, Mapping):
|
|
4519
|
+
return ""
|
|
4520
|
+
attributes = item.get("attributes") or item.get("attrs") or {}
|
|
4521
|
+
if isinstance(attributes, Sequence) and not isinstance(attributes, (str, bytes, bytearray)):
|
|
4522
|
+
flattened = {}
|
|
4523
|
+
for attr in attributes:
|
|
4524
|
+
if not isinstance(attr, Mapping):
|
|
4525
|
+
continue
|
|
4526
|
+
key = attr.get("key")
|
|
4527
|
+
value = attr.get("value")
|
|
4528
|
+
flattened[str(key)] = value
|
|
4529
|
+
attributes = flattened
|
|
4530
|
+
return _json_text(attributes)
|
|
4531
|
+
|
|
4532
|
+
|
|
4533
|
+
def _trace_coverage(
|
|
4534
|
+
trace_signals: Sequence[str],
|
|
4535
|
+
required_trace_signals: Sequence[str],
|
|
4536
|
+
) -> float:
|
|
4537
|
+
if not required_trace_signals:
|
|
4538
|
+
return 1.0
|
|
4539
|
+
present = set(trace_signals)
|
|
4540
|
+
required = set(required_trace_signals)
|
|
4541
|
+
return len(present & required) / len(required)
|
|
4542
|
+
|
|
4543
|
+
|
|
4544
|
+
def _has_transcript(record: Mapping[str, Any]) -> bool:
|
|
4545
|
+
text = _json_text(record)
|
|
4546
|
+
return any(
|
|
4547
|
+
token in text
|
|
4548
|
+
for token in (
|
|
4549
|
+
"transcript",
|
|
4550
|
+
"conversation_item",
|
|
4551
|
+
"user_input_transcribed",
|
|
4552
|
+
"assistant",
|
|
4553
|
+
"human",
|
|
4554
|
+
"message",
|
|
4555
|
+
)
|
|
4556
|
+
)
|
|
4557
|
+
|
|
4558
|
+
|
|
4559
|
+
def _has_error(record: Mapping[str, Any], trace_items: Sequence[Any]) -> bool:
|
|
4560
|
+
for item in [record, *trace_items]:
|
|
4561
|
+
if not isinstance(item, Mapping):
|
|
4562
|
+
continue
|
|
4563
|
+
if item.get("error") or item.get("exception") or item.get("error.type"):
|
|
4564
|
+
return True
|
|
4565
|
+
status = item.get("status")
|
|
4566
|
+
if isinstance(status, Mapping):
|
|
4567
|
+
code = str(status.get("code") or status.get("status_code") or "").lower()
|
|
4568
|
+
if code in {"error", "2"}:
|
|
4569
|
+
return True
|
|
4570
|
+
elif str(status or "").lower() in {"error", "failed", "failure"}:
|
|
4571
|
+
return True
|
|
4572
|
+
return False
|
|
4573
|
+
|
|
4574
|
+
|
|
4575
|
+
def _nested_get(value: Mapping[str, Any], path: Sequence[str]) -> Any:
|
|
4576
|
+
current: Any = value
|
|
4577
|
+
for part in path:
|
|
4578
|
+
if not isinstance(current, Mapping) or part not in current:
|
|
4579
|
+
return None
|
|
4580
|
+
current = current[part]
|
|
4581
|
+
return current
|
|
4582
|
+
|
|
4583
|
+
|
|
4584
|
+
def _first_string(record: Mapping[str, Any], *keys: str) -> Optional[str]:
|
|
4585
|
+
for key in keys:
|
|
4586
|
+
value = record.get(key)
|
|
4587
|
+
if value is not None:
|
|
4588
|
+
return str(value)
|
|
4589
|
+
metadata = record.get("metadata")
|
|
4590
|
+
if isinstance(metadata, Mapping):
|
|
4591
|
+
for key in keys:
|
|
4592
|
+
value = metadata.get(key)
|
|
4593
|
+
if value is not None:
|
|
4594
|
+
return str(value)
|
|
4595
|
+
return None
|
|
4596
|
+
|
|
4597
|
+
|
|
4598
|
+
def _candidate_id(
|
|
4599
|
+
record: Mapping[str, Any],
|
|
4600
|
+
candidate: Optional[AgentCandidate],
|
|
4601
|
+
) -> Optional[str]:
|
|
4602
|
+
return (
|
|
4603
|
+
_first_string(record, "candidate_id", "deployment_candidate_id")
|
|
4604
|
+
or (candidate.id if candidate is not None else None)
|
|
4605
|
+
)
|
|
4606
|
+
|
|
4607
|
+
|
|
4608
|
+
def _normalize_source(value: Any) -> str:
|
|
4609
|
+
normalized = str(value or "generic").strip().lower().replace("-", "_").replace(" ", "_")
|
|
4610
|
+
aliases = {
|
|
4611
|
+
"otel": "opentelemetry",
|
|
4612
|
+
"otlp": "opentelemetry",
|
|
4613
|
+
"traceai": "opentelemetry",
|
|
4614
|
+
"openai": "openai_agents",
|
|
4615
|
+
"openai_agent": "openai_agents",
|
|
4616
|
+
"livekit_agents": "livekit",
|
|
4617
|
+
}
|
|
4618
|
+
return aliases.get(normalized, normalized or "generic")
|
|
4619
|
+
|
|
4620
|
+
|
|
4621
|
+
def _normalize_signal(value: Any) -> str:
|
|
4622
|
+
normalized = str(value or "").strip().lower().replace("-", "_").replace(" ", "_")
|
|
4623
|
+
aliases = {
|
|
4624
|
+
"llm": "model",
|
|
4625
|
+
"function": "tool",
|
|
4626
|
+
"function_call": "tool",
|
|
4627
|
+
"messages": "message",
|
|
4628
|
+
"transcript": "message",
|
|
4629
|
+
}
|
|
4630
|
+
return aliases.get(normalized, normalized)
|
|
4631
|
+
|
|
4632
|
+
|
|
4633
|
+
def _json_text(value: Any) -> str:
|
|
4634
|
+
try:
|
|
4635
|
+
if hasattr(value, "model_dump"):
|
|
4636
|
+
value = value.model_dump()
|
|
4637
|
+
return json.dumps(value, sort_keys=True, default=str).lower()
|
|
4638
|
+
except Exception:
|
|
4639
|
+
return str(value).lower()
|