agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
fi/alk/bench/__init__.py
ADDED
|
@@ -0,0 +1,517 @@
|
|
|
1
|
+
"""Unified benchmark harness — the benchmarking front door to the simulation layer.
|
|
2
|
+
|
|
3
|
+
This module is a *composition facade*, not a new engine. A modern agent benchmark
|
|
4
|
+
harness decomposes into five layers — Task / Environment / Agent-adapter /
|
|
5
|
+
Verifier / Runner — and the kit already ships them (``tasks`` + worlds + the
|
|
6
|
+
framework adapters + ``evals``/``rewardhack`` + the live-lane runner + telemetry).
|
|
7
|
+
``bench`` adds the *contract glue* on top:
|
|
8
|
+
|
|
9
|
+
* **Fixed Task<->Verifier coupling** — a suite carries (or references) its own
|
|
10
|
+
oracle. This is the one part that never varies by modality.
|
|
11
|
+
* **Pluggable Environment + Agent-adapter** — the modality (text / tool / coding
|
|
12
|
+
/ voice / ...) is a dimension, not a fork.
|
|
13
|
+
* **Three control modes** —
|
|
14
|
+
- ``push`` : the harness drives the agent (today's ``run_benchmark``);
|
|
15
|
+
live for text / tool task datasets;
|
|
16
|
+
- ``artifact_in`` : score a submitted artifact, no live agent; live for coding
|
|
17
|
+
bench suites (subprocess or opt-in Docker sandbox);
|
|
18
|
+
- ``pull`` : the agent drives a live environment via reset/step
|
|
19
|
+
(staged; raises ``NotImplementedError`` until it lands).
|
|
20
|
+
Unsupported (suite, control_mode) combinations raise ``BenchError``.
|
|
21
|
+
* **A unified ``Result``** ``{scalar, components, pass_fail, explanation}`` that
|
|
22
|
+
every modality's verdict projects into. ``pass_fail`` keys are modality-defined
|
|
23
|
+
(push -> ``{"verdict": bool}``; coding -> ``{check_name: bool}``; void -> ``{}``);
|
|
24
|
+
the portable cross-modality signal is the row-level ``verdict`` + ``result.scalar``.
|
|
25
|
+
|
|
26
|
+
Honesty primitives are preserved verbatim: every per-task row keeps its
|
|
27
|
+
``execution_class`` / ``evidence_class`` and the overclaim tripwire, and the
|
|
28
|
+
reward-hack detector still fails a gamed run.
|
|
29
|
+
"""
|
|
30
|
+
|
|
31
|
+
from __future__ import annotations
|
|
32
|
+
|
|
33
|
+
import json
|
|
34
|
+
from pathlib import Path
|
|
35
|
+
from typing import Any, Mapping
|
|
36
|
+
|
|
37
|
+
from .. import tasks
|
|
38
|
+
|
|
39
|
+
# Re-exported coding-suite helpers (public): callers should use these, not the
|
|
40
|
+
# private ``bench._coding`` module. Explicit ``as`` re-export marks them public
|
|
41
|
+
# without an ``__all__`` (which would implicitly privatise run_bench /
|
|
42
|
+
# run_bench_file / load_bench_suite / modality_for_world_kind / BenchError / ...).
|
|
43
|
+
from ._coding import load_coding_suite as load_coding_suite
|
|
44
|
+
from ._coding import reference_submission as reference_submission
|
|
45
|
+
|
|
46
|
+
BENCH_RESULT_KIND = "agent-learning.bench-result.v1"
|
|
47
|
+
|
|
48
|
+
#: The control modes a bench run can take. ``push`` (text/tool) and ``artifact_in``
|
|
49
|
+
#: (coding suites) are live; ``pull`` is staged and raises ``NotImplementedError``.
|
|
50
|
+
CONTROL_MODES = ("push", "artifact_in", "pull")
|
|
51
|
+
|
|
52
|
+
#: Code sandboxes for the ``artifact_in`` coding lane.
|
|
53
|
+
SANDBOXES = ("subprocess", "docker")
|
|
54
|
+
|
|
55
|
+
#: World-kind -> coarse modality label. Kept deliberately small; new worlds map
|
|
56
|
+
#: here as they become executable.
|
|
57
|
+
_WORLD_KIND_MODALITY = {
|
|
58
|
+
"conversation": "text",
|
|
59
|
+
"tool_api": "tool",
|
|
60
|
+
"code_exec": "coding",
|
|
61
|
+
"browser": "computer_use",
|
|
62
|
+
"computer_use": "computer_use",
|
|
63
|
+
"voice_telephony": "voice",
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
class BenchError(ValueError):
|
|
68
|
+
"""Raised for malformed bench suites or invalid harness arguments."""
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def modality_for_world_kind(world_kind: str) -> str:
|
|
72
|
+
"""Map a world kind to its coarse modality label (``unknown`` if unmapped)."""
|
|
73
|
+
|
|
74
|
+
return _WORLD_KIND_MODALITY.get(str(world_kind), "unknown")
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def load_bench_suite(suite: Mapping[str, Any] | str | Path) -> dict[str, Any]:
|
|
78
|
+
"""Resolve a bench suite to a compiled TaskDataset.
|
|
79
|
+
|
|
80
|
+
Accepts a path (loaded + compiled via :func:`tasks.load_task_dataset`) or an
|
|
81
|
+
already-compiled dataset mapping (returned as a shallow copy). A raw,
|
|
82
|
+
uncompiled mapping is compiled via :func:`tasks.compile_task_dataset` so the
|
|
83
|
+
Goodhart guards are enforced before any run.
|
|
84
|
+
"""
|
|
85
|
+
|
|
86
|
+
if isinstance(suite, (str, Path)):
|
|
87
|
+
return tasks.load_task_dataset(suite)
|
|
88
|
+
if isinstance(suite, Mapping):
|
|
89
|
+
# A compiled dataset is idempotent under compile; compiling here keeps the
|
|
90
|
+
# guard checks on the Task<->Verifier coupling no matter how it arrived.
|
|
91
|
+
return tasks.compile_task_dataset(suite)
|
|
92
|
+
raise BenchError(
|
|
93
|
+
f"suite must be a path or a dataset mapping, got {type(suite).__name__!r}"
|
|
94
|
+
)
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def _project_result(row: Mapping[str, Any]) -> dict[str, Any]:
|
|
98
|
+
"""Project a ``tasks.run_benchmark`` per-task row into the unified Result.
|
|
99
|
+
|
|
100
|
+
The superset shape is synthesised from the strongest external references
|
|
101
|
+
(a metric->number map plus a value+rationale verdict): scalar score,
|
|
102
|
+
per-metric components, pass/fail booleans, and a short explanation.
|
|
103
|
+
"""
|
|
104
|
+
|
|
105
|
+
metric_averages = row.get("metric_averages") or {}
|
|
106
|
+
components = {
|
|
107
|
+
str(k): float(v)
|
|
108
|
+
for k, v in metric_averages.items()
|
|
109
|
+
if isinstance(v, (int, float)) and not isinstance(v, bool)
|
|
110
|
+
}
|
|
111
|
+
scoring = row.get("scoring") or {}
|
|
112
|
+
return {
|
|
113
|
+
"scalar": row.get("score"),
|
|
114
|
+
"components": components,
|
|
115
|
+
"pass_fail": {"verdict": row.get("verdict") == "pass"},
|
|
116
|
+
"explanation": scoring.get("basis"),
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def _bench_row(row: Mapping[str, Any], *, control_mode: str) -> dict[str, Any]:
|
|
121
|
+
world_kind = str(row.get("world_kind") or "")
|
|
122
|
+
out: dict[str, Any] = {
|
|
123
|
+
"task_id": row.get("task_id"),
|
|
124
|
+
"modality": modality_for_world_kind(world_kind),
|
|
125
|
+
"world_kind": world_kind,
|
|
126
|
+
"control_mode": control_mode,
|
|
127
|
+
"result": _project_result(row),
|
|
128
|
+
"verdict": row.get("verdict"),
|
|
129
|
+
"execution_class": row.get("execution_class"),
|
|
130
|
+
"evidence_class": row.get("evidence_class"),
|
|
131
|
+
"overclaim": bool(row.get("overclaim", False)),
|
|
132
|
+
}
|
|
133
|
+
if "rewardhack" in row:
|
|
134
|
+
out["rewardhack"] = row["rewardhack"]
|
|
135
|
+
if "error" in row:
|
|
136
|
+
out["error"] = row["error"]
|
|
137
|
+
return out
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def _to_bench_result(result: Mapping[str, Any], *, control_mode: str) -> dict[str, Any]:
|
|
141
|
+
"""Re-badge an engine benchmark result under the unified bench contract."""
|
|
142
|
+
|
|
143
|
+
per_task = [
|
|
144
|
+
_bench_row(r, control_mode=control_mode)
|
|
145
|
+
for r in result.get("per_task", [])
|
|
146
|
+
]
|
|
147
|
+
out: dict[str, Any] = {
|
|
148
|
+
"kind": BENCH_RESULT_KIND,
|
|
149
|
+
"control_mode": control_mode,
|
|
150
|
+
"dataset_name": result.get("dataset_name"),
|
|
151
|
+
"dataset_version": result.get("dataset_version"),
|
|
152
|
+
"modalities": sorted({r["modality"] for r in per_task}),
|
|
153
|
+
"per_task": per_task,
|
|
154
|
+
"aggregate": result.get("aggregate"),
|
|
155
|
+
}
|
|
156
|
+
if "telemetry" in result:
|
|
157
|
+
out["telemetry"] = result["telemetry"]
|
|
158
|
+
return out
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def _read_suite_obj(
|
|
162
|
+
suite: Mapping[str, Any] | str | Path,
|
|
163
|
+
) -> tuple[dict[str, Any], Path | None]:
|
|
164
|
+
if isinstance(suite, (str, Path)):
|
|
165
|
+
path = Path(suite).expanduser()
|
|
166
|
+
return json.loads(path.read_text("utf-8")), path
|
|
167
|
+
if isinstance(suite, Mapping):
|
|
168
|
+
return dict(suite), None
|
|
169
|
+
raise BenchError(
|
|
170
|
+
f"suite must be a path or a mapping, got {type(suite).__name__!r}"
|
|
171
|
+
)
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
_LIVE_EVIDENCE_CLASSES = ("live_lane", "live_stressed")
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def _coding_aggregate(rows: list[dict[str, Any]]) -> dict[str, Any]:
|
|
178
|
+
n = len(rows)
|
|
179
|
+
passed = sum(1 for r in rows if r["verdict"] == "pass")
|
|
180
|
+
voids = sum(1 for r in rows if r["verdict"] == "void")
|
|
181
|
+
scored = n - voids # void rows (no submission / infra failure) never ran
|
|
182
|
+
scalars = [
|
|
183
|
+
r["result"]["scalar"]
|
|
184
|
+
for r in rows
|
|
185
|
+
if r["result"].get("scalar") is not None
|
|
186
|
+
]
|
|
187
|
+
evidence_classes = {r.get("evidence_class") for r in rows}
|
|
188
|
+
|
|
189
|
+
def _group(key: str) -> dict[str, dict[str, int]]:
|
|
190
|
+
out: dict[str, dict[str, int]] = {}
|
|
191
|
+
for r in rows:
|
|
192
|
+
g = out.setdefault(str(r.get(key)), {"count": 0, "passed": 0})
|
|
193
|
+
g["count"] += 1
|
|
194
|
+
if r["verdict"] == "pass":
|
|
195
|
+
g["passed"] += 1
|
|
196
|
+
return out
|
|
197
|
+
|
|
198
|
+
return {
|
|
199
|
+
"count": n,
|
|
200
|
+
"passed": passed,
|
|
201
|
+
"void": voids,
|
|
202
|
+
"scored": scored,
|
|
203
|
+
# pass_rate is over SCORED tasks, not all tasks — so an infra failure
|
|
204
|
+
# (e.g. no Docker daemon) that voids every row does NOT read as "0% passed".
|
|
205
|
+
"pass_rate": round(passed / scored, 6) if scored else 0.0,
|
|
206
|
+
"mean_score": round(sum(scalars) / len(scalars), 6) if scalars else 0.0,
|
|
207
|
+
# derived from the actual rows (works for coding, pull/RL, any modality).
|
|
208
|
+
"by_world_kind": _group("world_kind"),
|
|
209
|
+
"by_execution_class": _group("execution_class"),
|
|
210
|
+
"by_modality": _group("modality"),
|
|
211
|
+
# honesty rollup (same 4-key shape as the push aggregate). Not gate-read;
|
|
212
|
+
# row-level honesty is the enforcement primitive.
|
|
213
|
+
"honesty": {
|
|
214
|
+
"evidence_class": next(iter(evidence_classes)) if len(evidence_classes) == 1 else sorted(c for c in evidence_classes if c),
|
|
215
|
+
"fixture_only": all(r.get("evidence_class") not in _LIVE_EVIDENCE_CLASSES for r in rows),
|
|
216
|
+
"any_live": any(r.get("evidence_class") in _LIVE_EVIDENCE_CLASSES for r in rows),
|
|
217
|
+
"any_overclaim": any(bool(r.get("overclaim")) for r in rows),
|
|
218
|
+
},
|
|
219
|
+
}
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
def _assemble(
|
|
223
|
+
rows: list[dict[str, Any]],
|
|
224
|
+
*,
|
|
225
|
+
control_mode: str,
|
|
226
|
+
name: str | None,
|
|
227
|
+
version: str | None,
|
|
228
|
+
emit_telemetry: bool,
|
|
229
|
+
project_name: str | None,
|
|
230
|
+
) -> dict[str, Any]:
|
|
231
|
+
aggregate = _coding_aggregate(rows)
|
|
232
|
+
out: dict[str, Any] = {
|
|
233
|
+
"kind": BENCH_RESULT_KIND,
|
|
234
|
+
"control_mode": control_mode,
|
|
235
|
+
"dataset_name": name,
|
|
236
|
+
"dataset_version": version,
|
|
237
|
+
"modalities": sorted({r["modality"] for r in rows}),
|
|
238
|
+
"per_task": rows,
|
|
239
|
+
"aggregate": aggregate,
|
|
240
|
+
}
|
|
241
|
+
if emit_telemetry:
|
|
242
|
+
from ..telemetry import emit_run
|
|
243
|
+
|
|
244
|
+
summary = emit_run(
|
|
245
|
+
kind="bench",
|
|
246
|
+
name=name or "bench",
|
|
247
|
+
metrics={
|
|
248
|
+
"n_tasks": aggregate["count"],
|
|
249
|
+
"pass_rate": aggregate["pass_rate"],
|
|
250
|
+
"mean_score": aggregate["mean_score"],
|
|
251
|
+
},
|
|
252
|
+
verdict="pass" if aggregate["pass_rate"] >= 0.5 else "fail",
|
|
253
|
+
children=[
|
|
254
|
+
(
|
|
255
|
+
f"task:{r['task_id']}",
|
|
256
|
+
{"verdict": r.get("verdict"), "score": r["result"].get("scalar")},
|
|
257
|
+
)
|
|
258
|
+
for r in rows
|
|
259
|
+
],
|
|
260
|
+
project_name=project_name,
|
|
261
|
+
)
|
|
262
|
+
out["telemetry"] = summary.as_dict()
|
|
263
|
+
return out
|
|
264
|
+
|
|
265
|
+
|
|
266
|
+
def _pull_row(
|
|
267
|
+
task: Mapping[str, Any], verdict_obj: Mapping[str, Any], *, evidence_class: str
|
|
268
|
+
) -> dict[str, Any]:
|
|
269
|
+
result = dict(verdict_obj["result"])
|
|
270
|
+
raw = verdict_obj.get("raw") or {}
|
|
271
|
+
if raw.get("infra_error"): # unknown env / bad policy -> the lane never ran
|
|
272
|
+
return {
|
|
273
|
+
"task_id": str(task.get("id")), "modality": "rl", "world_kind": "env",
|
|
274
|
+
"control_mode": "pull", "result": result, "verdict": "void",
|
|
275
|
+
"execution_class": "executable", "evidence_class": evidence_class,
|
|
276
|
+
"overclaim": False, "error": result.get("explanation"), "raw": raw,
|
|
277
|
+
}
|
|
278
|
+
pf = result.get("pass_fail") or {}
|
|
279
|
+
verdict = "pass" if pf.get("goal_reached") else "fail"
|
|
280
|
+
return {
|
|
281
|
+
"task_id": str(task.get("id")),
|
|
282
|
+
"modality": "rl",
|
|
283
|
+
"world_kind": "env",
|
|
284
|
+
"control_mode": "pull",
|
|
285
|
+
"result": result,
|
|
286
|
+
"verdict": verdict,
|
|
287
|
+
"execution_class": "executable",
|
|
288
|
+
"evidence_class": evidence_class,
|
|
289
|
+
"overclaim": False,
|
|
290
|
+
"raw": raw,
|
|
291
|
+
}
|
|
292
|
+
|
|
293
|
+
|
|
294
|
+
def _voice_row(
|
|
295
|
+
task: Mapping[str, Any], verdict_obj: Mapping[str, Any], *, evidence_class: str
|
|
296
|
+
) -> dict[str, Any]:
|
|
297
|
+
result = dict(verdict_obj["result"])
|
|
298
|
+
pf = result.get("pass_fail") or {}
|
|
299
|
+
verdict = "pass" if pf.get("voice") else "fail"
|
|
300
|
+
return {
|
|
301
|
+
"task_id": str(task.get("id")),
|
|
302
|
+
"modality": "voice",
|
|
303
|
+
"world_kind": "voice_telephony",
|
|
304
|
+
"control_mode": "artifact_in",
|
|
305
|
+
"result": result,
|
|
306
|
+
"verdict": verdict,
|
|
307
|
+
"execution_class": "executable",
|
|
308
|
+
"evidence_class": evidence_class,
|
|
309
|
+
"overclaim": False,
|
|
310
|
+
"raw": verdict_obj.get("raw", {}),
|
|
311
|
+
}
|
|
312
|
+
|
|
313
|
+
|
|
314
|
+
def _voice_void_row(task: Mapping[str, Any], evidence_class: str) -> dict[str, Any]:
|
|
315
|
+
return {
|
|
316
|
+
"task_id": str(task.get("id")),
|
|
317
|
+
"modality": "voice",
|
|
318
|
+
"world_kind": "voice_telephony",
|
|
319
|
+
"control_mode": "artifact_in",
|
|
320
|
+
"result": {"scalar": None, "components": {}, "pass_fail": {},
|
|
321
|
+
"explanation": "no transcript submitted"},
|
|
322
|
+
"verdict": "void",
|
|
323
|
+
"execution_class": "executable",
|
|
324
|
+
"evidence_class": evidence_class,
|
|
325
|
+
"overclaim": False,
|
|
326
|
+
"error": "no transcript submitted",
|
|
327
|
+
}
|
|
328
|
+
|
|
329
|
+
|
|
330
|
+
def run_bench(
|
|
331
|
+
suite: Mapping[str, Any] | str | Path,
|
|
332
|
+
agent: Mapping[str, Any] | None = None,
|
|
333
|
+
*,
|
|
334
|
+
control_mode: str = "push",
|
|
335
|
+
submission: Mapping[str, Any] | None = None,
|
|
336
|
+
sandbox: str = "subprocess",
|
|
337
|
+
split: str | None = None,
|
|
338
|
+
max_tasks: int | None = None,
|
|
339
|
+
seed: int = 42,
|
|
340
|
+
evidence_class: str = "captured_fixture",
|
|
341
|
+
detect_reward_hacks: bool = True,
|
|
342
|
+
runner: Any = None,
|
|
343
|
+
emit_telemetry: bool = True,
|
|
344
|
+
project_name: str | None = None,
|
|
345
|
+
) -> dict[str, Any]:
|
|
346
|
+
"""Run a bench ``suite`` and return a unified ``agent-learning.bench-result.v1``.
|
|
347
|
+
|
|
348
|
+
``control_mode``:
|
|
349
|
+
* ``push`` (default) — the harness drives ``agent`` through a world;
|
|
350
|
+
delegates to :func:`tasks.run_benchmark` (live today for text / tool).
|
|
351
|
+
* ``artifact_in`` — score a ``submission`` (``{task_id: candidate}``) against
|
|
352
|
+
each task's held-out oracle, no live agent. Requires a coding bench suite
|
|
353
|
+
(``agent-learning.bench-suite.v1``). ``sandbox`` selects the executor:
|
|
354
|
+
``subprocess`` (default) or ``docker`` (15E, hardened isolation).
|
|
355
|
+
* ``pull`` — agent-driven reset/step loop over a live environment (15D).
|
|
356
|
+
|
|
357
|
+
Per-task rows carry the unified ``result`` plus the preserved honesty fields
|
|
358
|
+
(``execution_class`` / ``evidence_class`` / ``overclaim``).
|
|
359
|
+
"""
|
|
360
|
+
|
|
361
|
+
if control_mode not in CONTROL_MODES:
|
|
362
|
+
raise BenchError(
|
|
363
|
+
f"unknown control_mode {control_mode!r}; expected one of {CONTROL_MODES}"
|
|
364
|
+
)
|
|
365
|
+
|
|
366
|
+
obj, path = _read_suite_obj(suite)
|
|
367
|
+
|
|
368
|
+
from . import _coding
|
|
369
|
+
|
|
370
|
+
if _coding.is_bench_suite(obj) and str(obj.get("control")) == "pull":
|
|
371
|
+
# Pull / RL suite: the agent (a policy callable or {"type": reference|noop})
|
|
372
|
+
# drives a simulated environment via reset/step.
|
|
373
|
+
from . import _pull
|
|
374
|
+
|
|
375
|
+
if control_mode != "pull":
|
|
376
|
+
raise BenchError(
|
|
377
|
+
f"pull bench suites run under control_mode='pull', not {control_mode!r}"
|
|
378
|
+
)
|
|
379
|
+
if agent is None:
|
|
380
|
+
raise BenchError("pull mode requires an agent (a policy callable or spec)")
|
|
381
|
+
if evidence_class not in tasks._evidence_classes():
|
|
382
|
+
raise BenchError(f"unknown evidence_class {evidence_class!r}")
|
|
383
|
+
task_list = list(obj.get("tasks") or [])
|
|
384
|
+
if max_tasks is not None:
|
|
385
|
+
task_list = task_list[: max(0, int(max_tasks))]
|
|
386
|
+
rows = [
|
|
387
|
+
_pull_row(t, _pull.run_pull(t, agent), evidence_class=evidence_class)
|
|
388
|
+
for t in task_list
|
|
389
|
+
]
|
|
390
|
+
return _assemble(
|
|
391
|
+
rows, control_mode="pull", name=str(obj.get("name") or ""),
|
|
392
|
+
version=str(obj.get("version") or ""), emit_telemetry=emit_telemetry,
|
|
393
|
+
project_name=project_name,
|
|
394
|
+
)
|
|
395
|
+
|
|
396
|
+
if _coding.is_bench_suite(obj) and str(obj.get("control")) == "voice":
|
|
397
|
+
# Voice suite: submit-and-score a voice episode transcript (artifact_in
|
|
398
|
+
# semantics). submission = {task_id: dialogue}. Deterministic verifier.
|
|
399
|
+
from . import _voice
|
|
400
|
+
|
|
401
|
+
if control_mode != "artifact_in":
|
|
402
|
+
raise BenchError(
|
|
403
|
+
f"voice bench suites run under control_mode='artifact_in', not {control_mode!r}"
|
|
404
|
+
)
|
|
405
|
+
if submission is None:
|
|
406
|
+
raise BenchError("voice artifact_in requires submission={task_id: dialogue}")
|
|
407
|
+
if evidence_class not in tasks._evidence_classes():
|
|
408
|
+
raise BenchError(f"unknown evidence_class {evidence_class!r}")
|
|
409
|
+
task_list = list(obj.get("tasks") or [])
|
|
410
|
+
if max_tasks is not None:
|
|
411
|
+
task_list = task_list[: max(0, int(max_tasks))]
|
|
412
|
+
rows = []
|
|
413
|
+
for t in task_list:
|
|
414
|
+
tid = str(t.get("id"))
|
|
415
|
+
dialogue = submission.get(tid)
|
|
416
|
+
if dialogue is None:
|
|
417
|
+
rows.append(_voice_void_row(t, evidence_class))
|
|
418
|
+
continue
|
|
419
|
+
vo = _voice.score_voice_episode(
|
|
420
|
+
dialogue, budgets=t.get("budgets"), required_content=t.get("required_content"),
|
|
421
|
+
)
|
|
422
|
+
rows.append(_voice_row(t, vo, evidence_class=evidence_class))
|
|
423
|
+
return _assemble(
|
|
424
|
+
rows, control_mode="artifact_in", name=str(obj.get("name") or ""),
|
|
425
|
+
version=str(obj.get("version") or ""), emit_telemetry=emit_telemetry,
|
|
426
|
+
project_name=project_name,
|
|
427
|
+
)
|
|
428
|
+
|
|
429
|
+
if _coding.is_bench_suite(obj):
|
|
430
|
+
coding_suite = _coding.load_coding_suite(obj)
|
|
431
|
+
if control_mode != "artifact_in":
|
|
432
|
+
raise BenchError(
|
|
433
|
+
f"coding bench suites currently run under control_mode='artifact_in', "
|
|
434
|
+
f"not {control_mode!r}"
|
|
435
|
+
)
|
|
436
|
+
if submission is None:
|
|
437
|
+
raise BenchError(
|
|
438
|
+
"artifact_in requires submission={task_id: candidate_source}"
|
|
439
|
+
)
|
|
440
|
+
if sandbox not in SANDBOXES:
|
|
441
|
+
raise BenchError(
|
|
442
|
+
f"unknown sandbox {sandbox!r}; expected one of {SANDBOXES}"
|
|
443
|
+
)
|
|
444
|
+
if evidence_class not in tasks._evidence_classes():
|
|
445
|
+
raise BenchError(
|
|
446
|
+
f"unknown evidence_class {evidence_class!r}; expected one of "
|
|
447
|
+
f"{tuple(tasks._evidence_classes())}"
|
|
448
|
+
)
|
|
449
|
+
rows = _coding.run_coding_artifact_in(
|
|
450
|
+
coding_suite,
|
|
451
|
+
submission,
|
|
452
|
+
sandbox=sandbox,
|
|
453
|
+
evidence_class=evidence_class,
|
|
454
|
+
max_tasks=max_tasks,
|
|
455
|
+
)
|
|
456
|
+
return _assemble(
|
|
457
|
+
rows,
|
|
458
|
+
control_mode="artifact_in",
|
|
459
|
+
name=str(coding_suite.get("name") or ""),
|
|
460
|
+
version=str(coding_suite.get("version") or ""),
|
|
461
|
+
emit_telemetry=emit_telemetry,
|
|
462
|
+
project_name=project_name,
|
|
463
|
+
)
|
|
464
|
+
|
|
465
|
+
# --- task-dataset suites (text / tool worlds) ---
|
|
466
|
+
if control_mode == "artifact_in":
|
|
467
|
+
raise BenchError(
|
|
468
|
+
"artifact_in currently requires a coding bench suite "
|
|
469
|
+
"(agent-learning.bench-suite.v1)"
|
|
470
|
+
)
|
|
471
|
+
if control_mode == "pull":
|
|
472
|
+
raise NotImplementedError(
|
|
473
|
+
"control_mode='pull' (agent-driven reset/step over a live environment) "
|
|
474
|
+
"lands in bench step 15D"
|
|
475
|
+
)
|
|
476
|
+
# control_mode == "push"
|
|
477
|
+
if agent is None:
|
|
478
|
+
raise BenchError("push mode requires an agent")
|
|
479
|
+
compiled = tasks.load_task_dataset(path) if path is not None else tasks.compile_task_dataset(obj)
|
|
480
|
+
result = tasks.run_benchmark(
|
|
481
|
+
compiled,
|
|
482
|
+
agent,
|
|
483
|
+
split=split,
|
|
484
|
+
max_tasks=max_tasks,
|
|
485
|
+
seed=seed,
|
|
486
|
+
evidence_class=evidence_class,
|
|
487
|
+
detect_reward_hacks=detect_reward_hacks,
|
|
488
|
+
runner=runner,
|
|
489
|
+
emit_telemetry=emit_telemetry,
|
|
490
|
+
project_name=project_name,
|
|
491
|
+
)
|
|
492
|
+
return _to_bench_result(result, control_mode="push")
|
|
493
|
+
|
|
494
|
+
|
|
495
|
+
def run_bench_file(
|
|
496
|
+
suite_path: str | Path,
|
|
497
|
+
agent: Mapping[str, Any] | None = None,
|
|
498
|
+
*,
|
|
499
|
+
output_path: str | Path | None = None,
|
|
500
|
+
**kwargs: Any,
|
|
501
|
+
) -> dict[str, Any]:
|
|
502
|
+
"""Convenience: :func:`run_bench` from a suite file, optionally writing JSON.
|
|
503
|
+
|
|
504
|
+
``agent`` is optional, mirroring :func:`run_bench`: it is required for
|
|
505
|
+
``push`` but unused for ``artifact_in`` (which takes ``submission=`` via
|
|
506
|
+
``kwargs``).
|
|
507
|
+
"""
|
|
508
|
+
|
|
509
|
+
payload = run_bench(Path(suite_path), agent, **kwargs)
|
|
510
|
+
if output_path is not None:
|
|
511
|
+
out = Path(output_path).expanduser()
|
|
512
|
+
out.parent.mkdir(parents=True, exist_ok=True)
|
|
513
|
+
out.write_text(
|
|
514
|
+
json.dumps(payload, indent=2, sort_keys=True, default=str) + "\n",
|
|
515
|
+
encoding="utf-8",
|
|
516
|
+
)
|
|
517
|
+
return payload
|