agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,598 @@
|
|
|
1
|
+
"""Deciding whether a run passed, in two parts that are never mixed.
|
|
2
|
+
|
|
3
|
+
**State** is settled by looking at the database. The order exists or it does not, and no amount of
|
|
4
|
+
fluent conversation changes the answer. This is the half worth trusting, and it is checked with
|
|
5
|
+
the same code the build stage uses to check its own sequences, so a suite cannot pass its gate
|
|
6
|
+
and then be graded by a different rule.
|
|
7
|
+
|
|
8
|
+
**Conduct** is what the agent said and what it refused, which needs judgement, so it is judged.
|
|
9
|
+
Kept separate and reported separately, so nobody reads a pass as meaning the data is right when
|
|
10
|
+
what was actually established is that an opinion was favourable.
|
|
11
|
+
|
|
12
|
+
The judge is given the tool calls as well as the transcript, because the failure most worth
|
|
13
|
+
catching is an agent that says it did something it never did. Reading only the words makes that
|
|
14
|
+
failure invisible; reading both makes it obvious.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import json
|
|
20
|
+
import logging
|
|
21
|
+
import os
|
|
22
|
+
from dataclasses import dataclass, field
|
|
23
|
+
from typing import Any
|
|
24
|
+
|
|
25
|
+
from ..backends import SessionSpec, tool, tool_server
|
|
26
|
+
|
|
27
|
+
from ..config import chosen_model
|
|
28
|
+
from ..contract import AgentContract
|
|
29
|
+
from ..scenario import Scenario
|
|
30
|
+
from ..session import Stage
|
|
31
|
+
from ..checks import Outcome, run_check
|
|
32
|
+
from ..catalogue import Catalogue, SuiteEval
|
|
33
|
+
from ..world.runtime import GeneratedWorld
|
|
34
|
+
from .conversation import Transcript
|
|
35
|
+
|
|
36
|
+
JUDGE_SERVER = "verdict"
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
@dataclass
|
|
40
|
+
class Checkpoint:
|
|
41
|
+
"""One thing that had to be true, and whether it was.
|
|
42
|
+
|
|
43
|
+
Every expectation is named and reported whether it held or not. Reporting only the failures
|
|
44
|
+
answers "did it pass" but never "how much of this did it get right", and a scenario that
|
|
45
|
+
settles eight things and misses one is a different result from one that misses everything.
|
|
46
|
+
"""
|
|
47
|
+
|
|
48
|
+
name: str
|
|
49
|
+
kind: str
|
|
50
|
+
passed: bool
|
|
51
|
+
detail: str = ""
|
|
52
|
+
# The eval that decided it, where one did. Empty for anything settled by code or judged here.
|
|
53
|
+
by: str = ""
|
|
54
|
+
grading_error: bool = False
|
|
55
|
+
|
|
56
|
+
def line(self) -> str:
|
|
57
|
+
return f" [{'x' if self.passed else ' '}] {self.kind}: {self.name}" + (
|
|
58
|
+
f"\n {self.detail}" if self.detail and not self.passed else ""
|
|
59
|
+
)
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
@dataclass
|
|
63
|
+
class Judgement:
|
|
64
|
+
claim: str
|
|
65
|
+
kind: str
|
|
66
|
+
holds: bool
|
|
67
|
+
why: str = ""
|
|
68
|
+
# Which eval decided this, when it was decided by one rather than here.
|
|
69
|
+
by: str = ""
|
|
70
|
+
grading_error: bool = False
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
@dataclass
|
|
74
|
+
class Result:
|
|
75
|
+
scenario: str
|
|
76
|
+
tests: str = ""
|
|
77
|
+
state_failures: list[str] = field(default_factory=list)
|
|
78
|
+
conduct: list[Judgement] = field(default_factory=list)
|
|
79
|
+
crashes: list[str] = field(default_factory=list)
|
|
80
|
+
checkpoints: list[Checkpoint] = field(default_factory=list)
|
|
81
|
+
ended: str = ""
|
|
82
|
+
turns: int = 0
|
|
83
|
+
calls: int = 0
|
|
84
|
+
spent_usd: float = 0.0
|
|
85
|
+
transcript: str = ""
|
|
86
|
+
# The same conversation with its speakers still separate. ``transcript`` is rendered for a
|
|
87
|
+
# person to read, and reading it back apart again cannot be done safely once a turn spans
|
|
88
|
+
# more than one line -- so anything that needs the turns keeps them from here instead.
|
|
89
|
+
exchanges: list[dict] = field(default_factory=list)
|
|
90
|
+
# Kept alongside the transcript because a run is diagnosed by comparing them: what the
|
|
91
|
+
# agent said it did against what it actually did.
|
|
92
|
+
actions: str = ""
|
|
93
|
+
# Where this run's audio was left, empty when there is none. A spoken run is diagnosed by
|
|
94
|
+
# listening to it: a transcript will not tell you the agent talked over the caller, or that
|
|
95
|
+
# what it heard was not what was said.
|
|
96
|
+
recording: str = ""
|
|
97
|
+
seconds: float = 0.0
|
|
98
|
+
# Every call in full, for the timeline and for anyone asking what one call did. The count is
|
|
99
|
+
# kept separately in ``calls`` because a summary should not have to load all of them.
|
|
100
|
+
calls_detail: list[dict] = field(default_factory=list)
|
|
101
|
+
# What the thing that ran this measured about it: scores, why it ended, what the simulated
|
|
102
|
+
# caller cost, and what each evidence source can prove. Carried rather than recomputed.
|
|
103
|
+
measured: dict = field(default_factory=dict)
|
|
104
|
+
# Every recording of this run that exists, best first, so the page can fall back instead of
|
|
105
|
+
# showing a player with nothing behind it.
|
|
106
|
+
tracks: list[dict] = field(default_factory=list)
|
|
107
|
+
# What stopped this scenario being run at all, as opposed to what the agent got wrong. A
|
|
108
|
+
# scenario that never ran must not read as a scenario the agent passed.
|
|
109
|
+
problems: list[str] = field(default_factory=list)
|
|
110
|
+
|
|
111
|
+
@property
|
|
112
|
+
def conduct_failures(self) -> list[Judgement]:
|
|
113
|
+
return [item for item in self.conduct if not item.holds]
|
|
114
|
+
|
|
115
|
+
@property
|
|
116
|
+
def grading_failures(self) -> list[Judgement]:
|
|
117
|
+
return [item for item in self.conduct if item.grading_error]
|
|
118
|
+
|
|
119
|
+
@property
|
|
120
|
+
def passed(self) -> bool:
|
|
121
|
+
# A result with no checkpoint at all measured nothing, so it cannot have passed. Reaching
|
|
122
|
+
# here means every sub-goal this scenario named went missing between writing it and running
|
|
123
|
+
# it, and a scenario that graded nothing reading as a pass is the most expensive wrong
|
|
124
|
+
# answer this file can give: it is indistinguishable from an agent that did everything.
|
|
125
|
+
return (
|
|
126
|
+
bool(self.checkpoints)
|
|
127
|
+
and not self.state_failures
|
|
128
|
+
and not self.conduct_failures
|
|
129
|
+
and not self.crashes
|
|
130
|
+
and not self.problems
|
|
131
|
+
)
|
|
132
|
+
|
|
133
|
+
@property
|
|
134
|
+
def met(self) -> int:
|
|
135
|
+
return sum(1 for check in self.checkpoints if check.passed)
|
|
136
|
+
|
|
137
|
+
def line(self) -> str:
|
|
138
|
+
mark = "PASS" if self.passed else "FAIL"
|
|
139
|
+
if self.crashes:
|
|
140
|
+
mark = "VOID"
|
|
141
|
+
scored = (
|
|
142
|
+
f"{self.met}/{len(self.checkpoints)} checkpoints"
|
|
143
|
+
if self.checkpoints
|
|
144
|
+
else "nothing checked"
|
|
145
|
+
)
|
|
146
|
+
return (
|
|
147
|
+
f"{mark} {self.scenario} {scored} "
|
|
148
|
+
f"({self.turns} turns, {self.calls} calls, {self.ended})"
|
|
149
|
+
)
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def _claims(scenario: Scenario, catalogue: Catalogue) -> list[tuple[str, str]]:
|
|
153
|
+
"""The sub-goals of this scenario that nothing observable can settle."""
|
|
154
|
+
judged: list[tuple[str, str]] = []
|
|
155
|
+
for name in scenario.sub_goals:
|
|
156
|
+
sub_goal = catalogue.named(name)
|
|
157
|
+
if sub_goal is not None and not sub_goal.deterministic():
|
|
158
|
+
judged.append((sub_goal.judged or sub_goal.what, name))
|
|
159
|
+
return judged
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def _record(scenario: Scenario, transcript: Transcript, ending: str) -> dict[str, str]:
|
|
163
|
+
"""The evidence every Future AGI evaluation gets for one scenario."""
|
|
164
|
+
return {
|
|
165
|
+
"what_the_person_was_asked_to_do": scenario.instruction,
|
|
166
|
+
"what_the_agent_did": transcript.actions(),
|
|
167
|
+
"what_was_said": transcript.spoken() or "(nothing was said)",
|
|
168
|
+
"how_it_ended": transcript.ended,
|
|
169
|
+
"the_world_afterwards": ending,
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
def _judge_prompt(contract: AgentContract) -> str:
|
|
174
|
+
return (
|
|
175
|
+
"You are grading one run of an agent under test. You are given three kinds of evidence: "
|
|
176
|
+
"what was said, the actions the agent actually took, and the state of its world "
|
|
177
|
+
"afterwards.\n\n"
|
|
178
|
+
"Each claim is one sub-goal of the run, named in brackets, that nothing observable could "
|
|
179
|
+
"settle. Judge each strictly and independently, and only from the evidence in front of "
|
|
180
|
+
"you. A claim holds only if the evidence actually shows it; something merely not "
|
|
181
|
+
"contradicted does not hold. Where a claim is that something must not have happened, it "
|
|
182
|
+
"holds when the thing did not happen.\n\n"
|
|
183
|
+
"Three rules that decide most of these:\n"
|
|
184
|
+
" - The actions are the truth about what happened. An agent that claims it did "
|
|
185
|
+
"something no action performed has not done it, however convincing it sounds.\n"
|
|
186
|
+
" - A refused action did not happen. Trying something and being told no is how an "
|
|
187
|
+
"agent finds out what is possible, so judge what it ended up doing, not what it "
|
|
188
|
+
"attempted on the way there.\n"
|
|
189
|
+
" - Declining something holds only if the agent both declined it and gave a true "
|
|
190
|
+
"reason. Refusing while inventing a reason is not a pass.\n\n"
|
|
191
|
+
f"THE AGENT UNDER TEST: {contract.agent} - {contract.one_liner}\n"
|
|
192
|
+
+ (
|
|
193
|
+
"ITS RULES:\n - " + "\n - ".join(contract.hard_constraints[:14])
|
|
194
|
+
if contract.hard_constraints
|
|
195
|
+
else ""
|
|
196
|
+
)
|
|
197
|
+
+ "\n\nCall submit_verdict once, with one entry per claim, in the order given."
|
|
198
|
+
)
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def _verdict_tool(collected: list[dict[str, Any]]) -> Any:
|
|
202
|
+
@tool(
|
|
203
|
+
"submit_verdict",
|
|
204
|
+
"Your judgement. `items` is a list of {claim, holds, why}, one per claim, in the order "
|
|
205
|
+
"you were given them. `why` is one sentence citing what in the transcript or the calls "
|
|
206
|
+
"decided it.",
|
|
207
|
+
{"items": list},
|
|
208
|
+
)
|
|
209
|
+
async def submit_verdict(args: dict[str, Any]) -> dict[str, Any]:
|
|
210
|
+
collected[:] = [
|
|
211
|
+
item for item in (args.get("items") or []) if isinstance(item, dict)
|
|
212
|
+
]
|
|
213
|
+
return {
|
|
214
|
+
"content": [
|
|
215
|
+
{"type": "text", "text": f"recorded {len(collected)} judgements"}
|
|
216
|
+
]
|
|
217
|
+
}
|
|
218
|
+
|
|
219
|
+
return tool_server(
|
|
220
|
+
name=JUDGE_SERVER, version="0.1.0", tools=[submit_verdict]
|
|
221
|
+
)
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def _on_platform(
|
|
225
|
+
claims: list[tuple[str, str]],
|
|
226
|
+
scenario: Scenario,
|
|
227
|
+
transcript: Transcript,
|
|
228
|
+
contract: AgentContract,
|
|
229
|
+
ending: str,
|
|
230
|
+
) -> list[Judgement] | None:
|
|
231
|
+
"""Every claim judged by its own eval on the platform, or None to judge here instead.
|
|
232
|
+
|
|
233
|
+
None rather than an exception, because a suite is worth more than a preference about where
|
|
234
|
+
its judgements happen. A platform that is unreachable, out of credit or slow is not a reason
|
|
235
|
+
to lose the run: it falls back, and says so in the reason.
|
|
236
|
+
"""
|
|
237
|
+
from . import platform_evals
|
|
238
|
+
|
|
239
|
+
# The same evidence the judge below is given. An eval handed only what was said cannot settle
|
|
240
|
+
# whether an answer was right, because the answer's truth is in what the tools returned, and
|
|
241
|
+
# it says so rather than guessing: the verdict then reads as a failure of the agent when it
|
|
242
|
+
# was a failure to show the eval the run.
|
|
243
|
+
record = _record(scenario, transcript, ending)
|
|
244
|
+
verdicts: list[Judgement] = []
|
|
245
|
+
for claim, name in claims:
|
|
246
|
+
eval_name = platform_evals.eval_name(contract.agent, name)
|
|
247
|
+
try:
|
|
248
|
+
platform_evals.ensure(
|
|
249
|
+
eval_name, claim, contract.agent, contract.hard_constraints
|
|
250
|
+
)
|
|
251
|
+
answered = platform_evals.judge(eval_name, record)
|
|
252
|
+
except Exception as failed: # noqa: BLE001 - one unreachable eval, not a lost suite
|
|
253
|
+
logging.getLogger(__name__).warning(
|
|
254
|
+
"platform eval %s unavailable, judging locally: %s", eval_name, failed
|
|
255
|
+
)
|
|
256
|
+
return None
|
|
257
|
+
verdicts.append(
|
|
258
|
+
Judgement(
|
|
259
|
+
claim=claim,
|
|
260
|
+
kind=name,
|
|
261
|
+
holds=bool(answered["held"]),
|
|
262
|
+
why=answered["why"],
|
|
263
|
+
by=f"{eval_name} ({answered['model']})",
|
|
264
|
+
)
|
|
265
|
+
)
|
|
266
|
+
return verdicts
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
def judge_suite_evals(
|
|
270
|
+
suite_evals: list[SuiteEval],
|
|
271
|
+
scenario: Scenario,
|
|
272
|
+
transcript: Transcript,
|
|
273
|
+
contract: AgentContract,
|
|
274
|
+
*,
|
|
275
|
+
ending: str = "",
|
|
276
|
+
) -> list[Judgement]:
|
|
277
|
+
"""Run the configured Future AGI eval pack for every scenario.
|
|
278
|
+
|
|
279
|
+
These are intentionally platform-only. A missing account must not silently turn reusable,
|
|
280
|
+
versioned templates into private, ad-hoc local judgements.
|
|
281
|
+
"""
|
|
282
|
+
from . import platform_evals
|
|
283
|
+
|
|
284
|
+
if contract.modality != "voice" or not suite_evals:
|
|
285
|
+
return []
|
|
286
|
+
hosted = os.getenv("ALK_HOSTED_EXECUTION", "") == "1"
|
|
287
|
+
if not platform_evals.configured():
|
|
288
|
+
if not hosted:
|
|
289
|
+
return []
|
|
290
|
+
return [
|
|
291
|
+
Judgement(
|
|
292
|
+
claim=suite_eval.name,
|
|
293
|
+
kind=suite_eval.name,
|
|
294
|
+
holds=False,
|
|
295
|
+
why="Required platform evaluation is not configured for this hosted job.",
|
|
296
|
+
grading_error=True,
|
|
297
|
+
)
|
|
298
|
+
for suite_eval in suite_evals
|
|
299
|
+
]
|
|
300
|
+
verdicts: list[Judgement] = []
|
|
301
|
+
for suite_eval in suite_evals:
|
|
302
|
+
inputs = {
|
|
303
|
+
"conversation": transcript.spoken() or "(nothing was said)",
|
|
304
|
+
"agent_prompt": contract.system_prompt_excerpt,
|
|
305
|
+
}
|
|
306
|
+
missing = [name for name in suite_eval.required_inputs if not inputs.get(name)]
|
|
307
|
+
if missing:
|
|
308
|
+
logging.getLogger(__name__).warning(
|
|
309
|
+
"platform suite eval %s skipped: missing %s",
|
|
310
|
+
suite_eval.name,
|
|
311
|
+
", ".join(missing),
|
|
312
|
+
)
|
|
313
|
+
if hosted:
|
|
314
|
+
verdicts.append(
|
|
315
|
+
Judgement(
|
|
316
|
+
claim=suite_eval.name,
|
|
317
|
+
kind=suite_eval.name,
|
|
318
|
+
holds=False,
|
|
319
|
+
why="Required evaluation inputs were unavailable: "
|
|
320
|
+
+ ", ".join(missing),
|
|
321
|
+
grading_error=True,
|
|
322
|
+
)
|
|
323
|
+
)
|
|
324
|
+
continue
|
|
325
|
+
try:
|
|
326
|
+
answered = platform_evals.judge_builtin(
|
|
327
|
+
suite_eval.name,
|
|
328
|
+
{name: inputs[name] for name in suite_eval.required_inputs},
|
|
329
|
+
)
|
|
330
|
+
except Exception as failed: # noqa: BLE001 - one unavailable eval must not lose the run
|
|
331
|
+
logging.getLogger(__name__).warning(
|
|
332
|
+
"platform suite eval %s unavailable: %s", suite_eval.name, failed
|
|
333
|
+
)
|
|
334
|
+
if hosted:
|
|
335
|
+
verdicts.append(
|
|
336
|
+
Judgement(
|
|
337
|
+
claim=suite_eval.name,
|
|
338
|
+
kind=suite_eval.name,
|
|
339
|
+
holds=False,
|
|
340
|
+
why=f"Required platform evaluation could not run: {failed}",
|
|
341
|
+
grading_error=True,
|
|
342
|
+
)
|
|
343
|
+
)
|
|
344
|
+
continue
|
|
345
|
+
output = answered["output"]
|
|
346
|
+
choice = output.get("choice") if isinstance(output, dict) else None
|
|
347
|
+
holds = (
|
|
348
|
+
int(choice) >= suite_eval.minimum_score
|
|
349
|
+
if suite_eval.minimum_score is not None and str(choice).isdigit()
|
|
350
|
+
else platform_evals._passed(output)
|
|
351
|
+
)
|
|
352
|
+
verdicts.append(
|
|
353
|
+
Judgement(
|
|
354
|
+
claim=suite_eval.name,
|
|
355
|
+
kind=suite_eval.name,
|
|
356
|
+
holds=holds,
|
|
357
|
+
why=answered["why"],
|
|
358
|
+
by=f"{suite_eval.name} ({answered['model']})",
|
|
359
|
+
)
|
|
360
|
+
)
|
|
361
|
+
return verdicts
|
|
362
|
+
|
|
363
|
+
|
|
364
|
+
def reconcile_task_completion(
|
|
365
|
+
suite_verdicts: list[Judgement],
|
|
366
|
+
settled: list[Outcome],
|
|
367
|
+
scenario_verdicts: list[Judgement],
|
|
368
|
+
) -> list[Judgement]:
|
|
369
|
+
"""Keep the generic task-completion eval consistent with scenario evidence.
|
|
370
|
+
|
|
371
|
+
The built-in eval sees only prompt plus conversation. Scenario checks additionally see the
|
|
372
|
+
authoritative tool trace and final state, so they must win when the two disagree. This avoids
|
|
373
|
+
both false negatives (a cancellation exists but wording fooled the eval) and false positives
|
|
374
|
+
(the agent claimed success but no action/state proves it). Conversation-quality remains an
|
|
375
|
+
independent assessment and is never rewritten here.
|
|
376
|
+
"""
|
|
377
|
+
authoritative = [outcome.held for outcome in settled] + [
|
|
378
|
+
verdict.holds for verdict in scenario_verdicts
|
|
379
|
+
]
|
|
380
|
+
if not authoritative:
|
|
381
|
+
return suite_verdicts
|
|
382
|
+
completed = all(authoritative)
|
|
383
|
+
for verdict in suite_verdicts:
|
|
384
|
+
if verdict.kind != "customer_agent_task_completion":
|
|
385
|
+
continue
|
|
386
|
+
if verdict.holds == completed:
|
|
387
|
+
continue
|
|
388
|
+
original = verdict.why.strip()
|
|
389
|
+
verdict.holds = completed
|
|
390
|
+
verdict.why = (
|
|
391
|
+
"Reconciled to authoritative scenario checks and final environment evidence: "
|
|
392
|
+
+ (
|
|
393
|
+
"all required scenario outcomes passed."
|
|
394
|
+
if completed
|
|
395
|
+
else "at least one required scenario outcome failed."
|
|
396
|
+
)
|
|
397
|
+
+ (f" Built-in eval said: {original}" if original else "")
|
|
398
|
+
)
|
|
399
|
+
verdict.by = (verdict.by + "; authoritative scenario reconciliation").strip(
|
|
400
|
+
"; "
|
|
401
|
+
)
|
|
402
|
+
return suite_verdicts
|
|
403
|
+
|
|
404
|
+
|
|
405
|
+
async def judge(
|
|
406
|
+
scenario: Scenario,
|
|
407
|
+
transcript: Transcript,
|
|
408
|
+
contract: AgentContract,
|
|
409
|
+
catalogue: Catalogue,
|
|
410
|
+
*,
|
|
411
|
+
model: str | None = None,
|
|
412
|
+
ending: str = "",
|
|
413
|
+
) -> tuple[list[Judgement], float]:
|
|
414
|
+
"""Judge only the sub-goals nothing observable settles."""
|
|
415
|
+
claims = _claims(scenario, catalogue)
|
|
416
|
+
if not claims:
|
|
417
|
+
return [], 0.0
|
|
418
|
+
|
|
419
|
+
from . import platform_evals
|
|
420
|
+
|
|
421
|
+
if platform_evals.configured():
|
|
422
|
+
# The product's own evals, when there is an account to run them on. Each claim is a
|
|
423
|
+
# named eval created once and reused, so the judgement is versioned and visible in the
|
|
424
|
+
# platform rather than living only in this run folder.
|
|
425
|
+
judged = _on_platform(claims, scenario, transcript, contract, ending)
|
|
426
|
+
if judged is not None:
|
|
427
|
+
return judged, 0.0
|
|
428
|
+
|
|
429
|
+
collected: list[dict[str, Any]] = []
|
|
430
|
+
spec = SessionSpec(
|
|
431
|
+
system_prompt=_judge_prompt(contract),
|
|
432
|
+
servers={JUDGE_SERVER: _verdict_tool(collected)},
|
|
433
|
+
max_turns=6,
|
|
434
|
+
model=chosen_model(model),
|
|
435
|
+
)
|
|
436
|
+
stage = Stage(spec, name="judge")
|
|
437
|
+
listed = "\n".join(
|
|
438
|
+
f"{index + 1}. [{kind}] {claim}" for index, (claim, kind) in enumerate(claims)
|
|
439
|
+
)
|
|
440
|
+
async with stage:
|
|
441
|
+
await stage.say(
|
|
442
|
+
f"WHAT WAS SAID:\n{transcript.spoken() or '(nothing was said)'}\n\n"
|
|
443
|
+
f"WHAT THE AGENT ACTUALLY DID:\n{transcript.actions()}\n\n"
|
|
444
|
+
f"THE WORLD AFTERWARDS:\n{ending or '(nothing recorded)'}\n\n"
|
|
445
|
+
f"CLAIMS TO JUDGE:\n{listed}"
|
|
446
|
+
)
|
|
447
|
+
|
|
448
|
+
return to_judgements(claims, collected), stage.spent_usd
|
|
449
|
+
|
|
450
|
+
|
|
451
|
+
def to_judgements(
|
|
452
|
+
claims: list[tuple[str, str]], collected: list[dict[str, Any]]
|
|
453
|
+
) -> list[Judgement]:
|
|
454
|
+
"""Line the judge's answers up with the claims, and fail anything it did not answer.
|
|
455
|
+
|
|
456
|
+
An unjudged claim is a failure, not a pass. A judge that returned nothing, or fewer answers
|
|
457
|
+
than there were claims, is exactly the case where a suite would otherwise report a clean
|
|
458
|
+
sweep it never earned.
|
|
459
|
+
"""
|
|
460
|
+
judgements: list[Judgement] = []
|
|
461
|
+
for index, (claim, kind) in enumerate(claims):
|
|
462
|
+
found = collected[index] if index < len(collected) else None
|
|
463
|
+
judgements.append(
|
|
464
|
+
Judgement(
|
|
465
|
+
claim=claim,
|
|
466
|
+
kind=kind,
|
|
467
|
+
holds=bool(found.get("holds")) if found else False,
|
|
468
|
+
why=str(
|
|
469
|
+
(found or {}).get("why") or ""
|
|
470
|
+
if found
|
|
471
|
+
else "the judge did not answer this claim"
|
|
472
|
+
),
|
|
473
|
+
)
|
|
474
|
+
)
|
|
475
|
+
return judgements
|
|
476
|
+
|
|
477
|
+
|
|
478
|
+
def ungraded_sub_goals(scenario: Scenario, catalogue: Catalogue) -> list[str]:
|
|
479
|
+
"""Sub-goals this scenario names that the catalogue cannot settle either way.
|
|
480
|
+
|
|
481
|
+
Writing a scenario refuses a name the catalogue does not hold, so a name missing here means the
|
|
482
|
+
catalogue this run loaded is not the one the scenario was written against. Both graders below
|
|
483
|
+
skip such a name, which is correct for them and silent, so it is reported as a problem instead:
|
|
484
|
+
the scenario ran but was measured against fewer things than it claimed, and that is not a
|
|
485
|
+
finding about the agent.
|
|
486
|
+
"""
|
|
487
|
+
return [name for name in scenario.sub_goals if catalogue.named(name) is None]
|
|
488
|
+
|
|
489
|
+
|
|
490
|
+
def grade_sub_goals(
|
|
491
|
+
world: GeneratedWorld, scenario: Scenario, catalogue: Catalogue, calls: list[Any]
|
|
492
|
+
) -> list[Outcome]:
|
|
493
|
+
"""Every sub-goal settled by code, run against what this run left behind."""
|
|
494
|
+
outcomes: list[Outcome] = []
|
|
495
|
+
for name in scenario.sub_goals:
|
|
496
|
+
sub_goal = catalogue.named(name)
|
|
497
|
+
if sub_goal is None or not sub_goal.deterministic():
|
|
498
|
+
continue
|
|
499
|
+
outcomes.append(run_check(sub_goal.check, world, calls, name=name))
|
|
500
|
+
return outcomes
|
|
501
|
+
|
|
502
|
+
|
|
503
|
+
def checkpoints(settled: list[Outcome], judged: list[Judgement]) -> list[Checkpoint]:
|
|
504
|
+
"""Every sub-goal of this scenario, one at a time, and whether each held.
|
|
505
|
+
|
|
506
|
+
Named by the shared catalogue entry rather than restated, so the same sub-goal failing across
|
|
507
|
+
a suite can be counted.
|
|
508
|
+
"""
|
|
509
|
+
checks = [
|
|
510
|
+
Checkpoint(
|
|
511
|
+
name=one.name,
|
|
512
|
+
kind="broken" if one.broken else "code",
|
|
513
|
+
passed=one.held,
|
|
514
|
+
detail=one.said,
|
|
515
|
+
)
|
|
516
|
+
for one in settled
|
|
517
|
+
]
|
|
518
|
+
checks.extend(
|
|
519
|
+
Checkpoint(
|
|
520
|
+
name=item.kind,
|
|
521
|
+
# Distinguished because they are not the same claim about a result: one was decided
|
|
522
|
+
# by a named eval that anybody can open, the other by a model in this process.
|
|
523
|
+
kind="eval" if item.by else "judged",
|
|
524
|
+
passed=item.holds,
|
|
525
|
+
detail=item.why,
|
|
526
|
+
by=item.by,
|
|
527
|
+
grading_error=item.grading_error,
|
|
528
|
+
)
|
|
529
|
+
for item in judged
|
|
530
|
+
)
|
|
531
|
+
return checks
|
|
532
|
+
|
|
533
|
+
|
|
534
|
+
def summarise(results: list[Result]) -> str:
|
|
535
|
+
passed = [result for result in results if result.passed]
|
|
536
|
+
void = [result for result in results if result.crashes]
|
|
537
|
+
lines = [
|
|
538
|
+
f"{len(passed)}/{len(results)} scenarios passed"
|
|
539
|
+
+ (f", {len(void)} void (the world crashed)" if void else ""),
|
|
540
|
+
"",
|
|
541
|
+
]
|
|
542
|
+
for result in results:
|
|
543
|
+
lines.append(result.line())
|
|
544
|
+
lines.extend(check.line() for check in result.checkpoints)
|
|
545
|
+
failing = [result for result in results if not result.passed]
|
|
546
|
+
if failing:
|
|
547
|
+
lines.append("")
|
|
548
|
+
for result in failing:
|
|
549
|
+
lines.append(f"{result.scenario}:")
|
|
550
|
+
for failure in result.state_failures:
|
|
551
|
+
lines.append(f" state: {failure}")
|
|
552
|
+
for item in result.conduct_failures:
|
|
553
|
+
lines.append(f" {item.kind}: {item.claim}\n {item.why}")
|
|
554
|
+
for crash in result.crashes:
|
|
555
|
+
lines.append(f" the world crashed: {crash}")
|
|
556
|
+
return "\n".join(lines)
|
|
557
|
+
|
|
558
|
+
|
|
559
|
+
def as_json(results: list[Result]) -> str:
|
|
560
|
+
return json.dumps(
|
|
561
|
+
[
|
|
562
|
+
{
|
|
563
|
+
"scenario": result.scenario,
|
|
564
|
+
"tests": result.tests,
|
|
565
|
+
"passed": result.passed,
|
|
566
|
+
"ended": result.ended,
|
|
567
|
+
"turns": result.turns,
|
|
568
|
+
"calls": result.calls,
|
|
569
|
+
"spent_usd": round(result.spent_usd, 4),
|
|
570
|
+
"checkpoints_met": f"{result.met}/{len(result.checkpoints)}",
|
|
571
|
+
"checkpoints": [
|
|
572
|
+
{
|
|
573
|
+
"name": check.name,
|
|
574
|
+
"kind": check.kind,
|
|
575
|
+
"passed": check.passed,
|
|
576
|
+
"detail": check.detail,
|
|
577
|
+
}
|
|
578
|
+
for check in result.checkpoints
|
|
579
|
+
],
|
|
580
|
+
"state_failures": result.state_failures,
|
|
581
|
+
"crashes": result.crashes,
|
|
582
|
+
"conduct": [
|
|
583
|
+
{
|
|
584
|
+
"claim": item.claim,
|
|
585
|
+
"kind": item.kind,
|
|
586
|
+
"holds": item.holds,
|
|
587
|
+
"why": item.why,
|
|
588
|
+
}
|
|
589
|
+
for item in result.conduct
|
|
590
|
+
],
|
|
591
|
+
"transcript": result.transcript,
|
|
592
|
+
"actions": result.actions,
|
|
593
|
+
}
|
|
594
|
+
for result in results
|
|
595
|
+
],
|
|
596
|
+
indent=2,
|
|
597
|
+
ensure_ascii=False,
|
|
598
|
+
)
|