agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
fi/alk/bench/_grader.py
ADDED
|
@@ -0,0 +1,286 @@
|
|
|
1
|
+
"""Command/artifact-graded coding lane — the hardened coding tier.
|
|
2
|
+
|
|
3
|
+
This resolves the in-process forge/oracle-read weakness of the ``check_*`` lane by
|
|
4
|
+
changing the *model*, not bolting on isolation. A task gives the candidate a
|
|
5
|
+
working directory + a way to RUN it; a **held-out grader** runs AFTERWARD and
|
|
6
|
+
emits the verdict via its **exit code + a reward file in a grader-controlled
|
|
7
|
+
path** — never parsed from candidate-shared stdout.
|
|
8
|
+
|
|
9
|
+
Why this is robust (the two vulns from the PR review, structurally closed):
|
|
10
|
+
|
|
11
|
+
* **No verdict forgery** — the verdict is the grader's exit code (and an optional
|
|
12
|
+
``reward.json`` the grader writes), not anything the candidate prints.
|
|
13
|
+
* **No oracle read** — the grader files (held-out expected values / tests) are
|
|
14
|
+
written ONLY after the candidate command has finished and its processes are
|
|
15
|
+
killed. The candidate never co-runs with the grader, so it cannot read the
|
|
16
|
+
expected values; and in the Docker lane the grader files are owned by a
|
|
17
|
+
different user the candidate uid cannot read.
|
|
18
|
+
|
|
19
|
+
It also gives **multi-language for free**: the candidate ``build`` command and the
|
|
20
|
+
``grader`` command are arbitrary shell, so the same lane grades Python, bash,
|
|
21
|
+
Node, compiled languages, etc.
|
|
22
|
+
|
|
23
|
+
Two sandboxes share one flow (temporal separation candidate→grader):
|
|
24
|
+
* ``subprocess`` — candidate + grader run as host subprocesses in *separate*
|
|
25
|
+
temp dirs; the grader dir path is given only to the grader. Credential-free,
|
|
26
|
+
Docker-free; used by the release gate on trusted shipped tasks.
|
|
27
|
+
* ``docker`` — a per-task, network-off, capped, ephemeral container; candidate
|
|
28
|
+
runs as an unprivileged uid, grader files land in a root-owned dir the
|
|
29
|
+
candidate cannot read. The hardened lane for untrusted agent output.
|
|
30
|
+
"""
|
|
31
|
+
|
|
32
|
+
from __future__ import annotations
|
|
33
|
+
|
|
34
|
+
import json
|
|
35
|
+
import os
|
|
36
|
+
import subprocess
|
|
37
|
+
import tempfile
|
|
38
|
+
import uuid
|
|
39
|
+
from pathlib import Path
|
|
40
|
+
from typing import Any, Mapping
|
|
41
|
+
|
|
42
|
+
from ._codeexec import _empty_result, _tail
|
|
43
|
+
|
|
44
|
+
GRADING_COMMAND = "command" # suite/task grading mode discriminator
|
|
45
|
+
_DEFAULT_TIMEOUT_S = 20.0
|
|
46
|
+
_REWARD_FILE = "reward.json"
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _files(value: Any) -> dict[str, str]:
|
|
50
|
+
"""Coerce a {path: content} mapping to str->str (defensive)."""
|
|
51
|
+
|
|
52
|
+
if not isinstance(value, Mapping):
|
|
53
|
+
return {}
|
|
54
|
+
return {str(k): str(v) for k, v in value.items()}
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def _write_tree(root: Path, files: Mapping[str, str]) -> None:
|
|
58
|
+
for rel, content in files.items():
|
|
59
|
+
dest = root / rel
|
|
60
|
+
dest.parent.mkdir(parents=True, exist_ok=True)
|
|
61
|
+
dest.write_text(content, encoding="utf-8")
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _result_from_grader(
|
|
65
|
+
rc: int | None, reward: Mapping[str, Any] | None, raw: dict[str, Any]
|
|
66
|
+
) -> dict[str, Any]:
|
|
67
|
+
"""Build the unified Result from the grader's exit code + optional reward.json.
|
|
68
|
+
|
|
69
|
+
Verdict authority: the grader's exit code (0 = pass). If the grader also wrote
|
|
70
|
+
a ``reward.json`` with a numeric ``score`` in [0,1], that becomes the scalar;
|
|
71
|
+
otherwise the scalar is 1.0/0.0 from the exit code. Sub-check booleans, if the
|
|
72
|
+
grader reports a ``checks`` map, flow into ``pass_fail``.
|
|
73
|
+
"""
|
|
74
|
+
|
|
75
|
+
passed = rc == 0
|
|
76
|
+
scalar: float
|
|
77
|
+
pass_fail: dict[str, bool] = {}
|
|
78
|
+
explanation = f"grader exit {rc}"
|
|
79
|
+
if isinstance(reward, Mapping):
|
|
80
|
+
score = reward.get("score")
|
|
81
|
+
if isinstance(score, (int, float)) and not isinstance(score, bool):
|
|
82
|
+
scalar = round(float(score), 6)
|
|
83
|
+
else:
|
|
84
|
+
scalar = 1.0 if passed else 0.0
|
|
85
|
+
checks = reward.get("checks")
|
|
86
|
+
if isinstance(checks, Mapping):
|
|
87
|
+
pass_fail = {str(k): bool(v) for k, v in checks.items()}
|
|
88
|
+
if reward.get("explanation"):
|
|
89
|
+
explanation = str(reward["explanation"])
|
|
90
|
+
else:
|
|
91
|
+
scalar = 1.0 if passed else 0.0
|
|
92
|
+
if not pass_fail:
|
|
93
|
+
pass_fail = {"grader": passed}
|
|
94
|
+
return {
|
|
95
|
+
"result": {
|
|
96
|
+
"scalar": scalar,
|
|
97
|
+
"components": {"grader_exit_ok": 1.0 if passed else 0.0},
|
|
98
|
+
"pass_fail": pass_fail,
|
|
99
|
+
"explanation": explanation,
|
|
100
|
+
},
|
|
101
|
+
"raw": raw,
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def run_command_graded(
|
|
106
|
+
task: Mapping[str, Any],
|
|
107
|
+
candidate_files: Mapping[str, str],
|
|
108
|
+
*,
|
|
109
|
+
sandbox: str = "subprocess",
|
|
110
|
+
timeout_s: float = _DEFAULT_TIMEOUT_S,
|
|
111
|
+
) -> dict[str, Any]:
|
|
112
|
+
"""Grade ``candidate_files`` for a command-graded ``task``.
|
|
113
|
+
|
|
114
|
+
Returns ``{"result", "raw"}``. Never raises for infra problems (missing Docker,
|
|
115
|
+
grader crash) — those surface as a failing/infra Result, tagged
|
|
116
|
+
``raw["infra_error"]`` when the lane could not run at all.
|
|
117
|
+
"""
|
|
118
|
+
|
|
119
|
+
if sandbox == "docker":
|
|
120
|
+
from ._docker import docker_available
|
|
121
|
+
|
|
122
|
+
if not docker_available():
|
|
123
|
+
return _empty_result(
|
|
124
|
+
"docker unavailable (no daemon / not installed)",
|
|
125
|
+
{"sandbox": "docker", "grading": GRADING_COMMAND, "infra_error": True},
|
|
126
|
+
)
|
|
127
|
+
return _run_docker_graded(task, candidate_files, timeout_s=timeout_s)
|
|
128
|
+
if sandbox != "subprocess":
|
|
129
|
+
return _empty_result(
|
|
130
|
+
f"unknown sandbox {sandbox!r}; expected 'subprocess' or 'docker'",
|
|
131
|
+
{"sandbox": sandbox, "grading": GRADING_COMMAND, "infra_error": True},
|
|
132
|
+
)
|
|
133
|
+
return _run_subprocess_graded(task, candidate_files, timeout_s=timeout_s)
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def _run_subprocess_graded(
|
|
137
|
+
task: Mapping[str, Any], candidate_files: Mapping[str, str], *, timeout_s: float
|
|
138
|
+
) -> dict[str, Any]:
|
|
139
|
+
build = task.get("build")
|
|
140
|
+
grader_cmd = str(task.get("grader_cmd") or "")
|
|
141
|
+
if not grader_cmd:
|
|
142
|
+
return _empty_result("task has no grader_cmd", {"sandbox": "subprocess", "infra_error": True})
|
|
143
|
+
|
|
144
|
+
raw: dict[str, Any] = {"sandbox": "subprocess", "grading": GRADING_COMMAND, "timed_out": False}
|
|
145
|
+
with tempfile.TemporaryDirectory(prefix="bench-work-") as work_s, \
|
|
146
|
+
tempfile.TemporaryDirectory(prefix="bench-grader-") as grader_s:
|
|
147
|
+
work = Path(work_s)
|
|
148
|
+
grader = Path(grader_s)
|
|
149
|
+
_write_tree(work, {**_files(task.get("files")), **dict(candidate_files)})
|
|
150
|
+
|
|
151
|
+
# PHASE 1 — candidate runs with NO grader present (temporal hold-out).
|
|
152
|
+
if build:
|
|
153
|
+
try:
|
|
154
|
+
proc = subprocess.run(
|
|
155
|
+
["sh", "-c", str(build)], cwd=str(work), capture_output=True,
|
|
156
|
+
text=True, timeout=timeout_s,
|
|
157
|
+
)
|
|
158
|
+
raw["build_exit"] = proc.returncode
|
|
159
|
+
raw["build_stdout_tail"] = _tail(proc.stdout)
|
|
160
|
+
except subprocess.TimeoutExpired:
|
|
161
|
+
raw["timed_out"] = True
|
|
162
|
+
return _empty_result(f"candidate build timed out after {timeout_s}s", raw)
|
|
163
|
+
|
|
164
|
+
# PHASE 2 — grader written AFTER, in a dir the candidate phase never knew.
|
|
165
|
+
_write_tree(grader, _files(task.get("grader_files")))
|
|
166
|
+
# The subprocess lane is the trusted/gate tier (not a security boundary —
|
|
167
|
+
# the Docker lane is), so the grader gets a usable PATH to resolve
|
|
168
|
+
# interpreters; GRADER_DIR points it at its held-out files.
|
|
169
|
+
env = {
|
|
170
|
+
"PATH": os.environ.get("PATH", "/usr/bin:/bin:/usr/local/bin"),
|
|
171
|
+
"GRADER_DIR": str(grader),
|
|
172
|
+
"HOME": os.environ.get("HOME", str(work)),
|
|
173
|
+
}
|
|
174
|
+
try:
|
|
175
|
+
gproc = subprocess.run(
|
|
176
|
+
["sh", "-c", grader_cmd], cwd=str(work), capture_output=True,
|
|
177
|
+
text=True, timeout=timeout_s, env=env,
|
|
178
|
+
)
|
|
179
|
+
except subprocess.TimeoutExpired:
|
|
180
|
+
raw["timed_out"] = True
|
|
181
|
+
return _empty_result(f"grader timed out after {timeout_s}s", raw)
|
|
182
|
+
raw["grader_exit"] = gproc.returncode
|
|
183
|
+
raw["grader_stdout_tail"] = _tail(gproc.stdout)
|
|
184
|
+
raw["grader_stderr_tail"] = _tail(gproc.stderr)
|
|
185
|
+
|
|
186
|
+
reward = _read_reward(grader / _REWARD_FILE)
|
|
187
|
+
return _result_from_grader(gproc.returncode, reward, raw)
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def _read_reward(path: Path) -> dict[str, Any] | None:
|
|
191
|
+
try:
|
|
192
|
+
if path.exists():
|
|
193
|
+
return json.loads(path.read_text(encoding="utf-8"))
|
|
194
|
+
except (OSError, json.JSONDecodeError):
|
|
195
|
+
return None
|
|
196
|
+
return None
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
# ---- Docker command-graded lane (hardened, opt-in) ----
|
|
200
|
+
|
|
201
|
+
_DOCKER_IMAGE = "python:3.11-slim"
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def _docker(*args: str, timeout: float = 30.0) -> subprocess.CompletedProcess:
|
|
205
|
+
return subprocess.run(
|
|
206
|
+
["docker", *args], capture_output=True, text=True, timeout=timeout
|
|
207
|
+
)
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
def _run_docker_graded(
|
|
211
|
+
task: Mapping[str, Any], candidate_files: Mapping[str, str], *, timeout_s: float
|
|
212
|
+
) -> dict[str, Any]:
|
|
213
|
+
build = task.get("build")
|
|
214
|
+
grader_cmd = str(task.get("grader_cmd") or "")
|
|
215
|
+
image = str(task.get("image") or _DOCKER_IMAGE)
|
|
216
|
+
if not grader_cmd:
|
|
217
|
+
return _empty_result("task has no grader_cmd", {"sandbox": "docker", "infra_error": True})
|
|
218
|
+
|
|
219
|
+
name = f"agent-learn-grade-{uuid.uuid4().hex[:12]}"
|
|
220
|
+
raw: dict[str, Any] = {
|
|
221
|
+
"sandbox": "docker", "grading": GRADING_COMMAND, "image": image,
|
|
222
|
+
"network": "none", "container": name, "timed_out": False,
|
|
223
|
+
}
|
|
224
|
+
with tempfile.TemporaryDirectory(prefix="bench-docker-") as host_s:
|
|
225
|
+
host = Path(host_s)
|
|
226
|
+
work_host = host / "work"
|
|
227
|
+
grader_host = host / "grader"
|
|
228
|
+
_write_tree(work_host, {**_files(task.get("files")), **dict(candidate_files)})
|
|
229
|
+
_write_tree(grader_host, _files(task.get("grader_files")))
|
|
230
|
+
|
|
231
|
+
started = _docker(
|
|
232
|
+
"run", "-d", "--rm", "--name", name, "--network", "none",
|
|
233
|
+
"--memory", "512m", "--cpus", "1.0", "--pids-limit", "256",
|
|
234
|
+
"--cap-drop", "ALL", "--security-opt", "no-new-privileges",
|
|
235
|
+
image, "sleep", str(int(timeout_s * 2 + 60)),
|
|
236
|
+
)
|
|
237
|
+
if started.returncode != 0:
|
|
238
|
+
raw["infra_error"] = True
|
|
239
|
+
return _empty_result(
|
|
240
|
+
f"docker run failed: {_tail(started.stderr, 300)}", raw
|
|
241
|
+
)
|
|
242
|
+
try:
|
|
243
|
+
# candidate user + dirs; /work candidate-writable, /grader root-only.
|
|
244
|
+
_docker("exec", name, "sh", "-c",
|
|
245
|
+
"id cand 2>/dev/null || useradd -M -s /usr/sbin/nologin cand; "
|
|
246
|
+
"mkdir -p /work /grader")
|
|
247
|
+
_docker("cp", f"{work_host}/.", f"{name}:/work")
|
|
248
|
+
_docker("exec", name, "sh", "-c", "chown -R cand:cand /work && chmod 700 /grader")
|
|
249
|
+
|
|
250
|
+
# PHASE 1 — candidate runs as `cand`, no grader files present yet.
|
|
251
|
+
if build:
|
|
252
|
+
try:
|
|
253
|
+
b = _docker("exec", "-u", "cand", "-w", "/work", name,
|
|
254
|
+
"sh", "-c", str(build), timeout=timeout_s + 15)
|
|
255
|
+
raw["build_exit"] = b.returncode
|
|
256
|
+
raw["build_stdout_tail"] = _tail(b.stdout)
|
|
257
|
+
except subprocess.TimeoutExpired:
|
|
258
|
+
raw["timed_out"] = True
|
|
259
|
+
return _empty_result(f"candidate build timed out after {timeout_s}s", raw)
|
|
260
|
+
|
|
261
|
+
# kill any lingering candidate processes before grading.
|
|
262
|
+
_docker("exec", name, "sh", "-c", "pkill -u cand 2>/dev/null || true")
|
|
263
|
+
|
|
264
|
+
# PHASE 2 — inject grader (root-owned, unreadable to cand), run as root.
|
|
265
|
+
_docker("cp", f"{grader_host}/.", f"{name}:/grader")
|
|
266
|
+
_docker("exec", name, "sh", "-c", "chown -R root:root /grader && chmod -R go-rwx /grader")
|
|
267
|
+
try:
|
|
268
|
+
g = _docker("exec", "-w", "/work", name, "sh", "-c",
|
|
269
|
+
f"export GRADER_DIR=/grader; {grader_cmd}", timeout=timeout_s + 15)
|
|
270
|
+
except subprocess.TimeoutExpired:
|
|
271
|
+
raw["timed_out"] = True
|
|
272
|
+
return _empty_result(f"grader timed out after {timeout_s}s", raw)
|
|
273
|
+
raw["grader_exit"] = g.returncode
|
|
274
|
+
raw["grader_stdout_tail"] = _tail(g.stdout)
|
|
275
|
+
raw["grader_stderr_tail"] = _tail(g.stderr)
|
|
276
|
+
|
|
277
|
+
cat = _docker("exec", name, "sh", "-c", "cat /grader/reward.json 2>/dev/null || true")
|
|
278
|
+
reward: dict[str, Any] | None = None
|
|
279
|
+
if cat.stdout.strip():
|
|
280
|
+
try:
|
|
281
|
+
reward = json.loads(cat.stdout)
|
|
282
|
+
except json.JSONDecodeError:
|
|
283
|
+
reward = None
|
|
284
|
+
return _result_from_grader(g.returncode, reward, raw)
|
|
285
|
+
finally:
|
|
286
|
+
_docker("rm", "-f", name)
|
fi/alk/bench/_pull.py
ADDED
|
@@ -0,0 +1,212 @@
|
|
|
1
|
+
"""Pull / RL control mode — the AGENT drives a live environment via reset/step.
|
|
2
|
+
|
|
3
|
+
The push lane has the harness drive the agent; artifact-in scores a submitted
|
|
4
|
+
artifact. **Pull** inverts control: the agent is a policy ``obs -> action`` that
|
|
5
|
+
steps an environment until done, and the score is the environment's reward. This
|
|
6
|
+
is the Gym/OpenEnv shape, run live (not replayed).
|
|
7
|
+
|
|
8
|
+
Deep-contract + simulated: the environments here are deterministic, in-process,
|
|
9
|
+
credential-free simulators (so the lane is fully gate-verifiable). A *live*
|
|
10
|
+
external env server (an HTTP step/reset endpoint) is the same contract with a
|
|
11
|
+
network transport and is deferred to owner infra — it plugs in as another
|
|
12
|
+
``Environment`` without changing the driver or the unified Result.
|
|
13
|
+
|
|
14
|
+
An environment implements:
|
|
15
|
+
* ``reset(spec) -> (state, obs)``
|
|
16
|
+
* ``step(state, action) -> (state, obs, reward, done, info)``
|
|
17
|
+
* ``optimal_action(obs) -> action`` — a reference policy (proves solvability)
|
|
18
|
+
* ``actions`` — the discrete action set
|
|
19
|
+
|
|
20
|
+
A policy is a callable ``obs -> action`` or a spec dict: ``{"type": "reference"}``
|
|
21
|
+
(the env's optimal policy) or ``{"type": "noop"}`` (always the first action).
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
from __future__ import annotations
|
|
25
|
+
|
|
26
|
+
from typing import Any, Callable, Mapping, Protocol
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class Environment(Protocol):
|
|
30
|
+
actions: tuple[str, ...]
|
|
31
|
+
|
|
32
|
+
def reset(self, spec: Mapping[str, Any]) -> tuple[dict, dict]: ...
|
|
33
|
+
def step(self, state: dict, action: str) -> tuple[dict, dict, float, bool, dict]: ...
|
|
34
|
+
def optimal_action(self, obs: Mapping[str, Any]) -> str: ...
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
class ReachTargetEnv:
|
|
38
|
+
"""1-D navigation: move toward ``target`` from ``start`` within ``max_steps``.
|
|
39
|
+
|
|
40
|
+
obs = {pos, target, remaining}. Reward 1.0 the step the agent lands on target
|
|
41
|
+
(then done); 0.0 otherwise. Deterministic + trivially verifiable; the optimal
|
|
42
|
+
policy is "step toward target".
|
|
43
|
+
"""
|
|
44
|
+
|
|
45
|
+
actions: tuple[str, ...] = ("left", "right", "stay")
|
|
46
|
+
|
|
47
|
+
def reset(self, spec: Mapping[str, Any]) -> tuple[dict, dict]:
|
|
48
|
+
state = {
|
|
49
|
+
"pos": int(spec.get("start", 0)),
|
|
50
|
+
"target": int(spec.get("target", 5)),
|
|
51
|
+
"steps": 0,
|
|
52
|
+
"max_steps": int(spec.get("max_steps", 20)),
|
|
53
|
+
}
|
|
54
|
+
return state, self._obs(state)
|
|
55
|
+
|
|
56
|
+
def step(self, state: dict, action: str) -> tuple[dict, dict, float, bool, dict]:
|
|
57
|
+
state = dict(state)
|
|
58
|
+
state["pos"] += {"left": -1, "right": 1, "stay": 0}.get(action, 0)
|
|
59
|
+
state["steps"] += 1
|
|
60
|
+
reached = state["pos"] == state["target"]
|
|
61
|
+
done = reached or state["steps"] >= state["max_steps"]
|
|
62
|
+
reward = 1.0 if reached else 0.0
|
|
63
|
+
return state, self._obs(state), reward, done, {"reached": reached}
|
|
64
|
+
|
|
65
|
+
def optimal_action(self, obs: Mapping[str, Any]) -> str:
|
|
66
|
+
if obs["pos"] < obs["target"]:
|
|
67
|
+
return "right"
|
|
68
|
+
if obs["pos"] > obs["target"]:
|
|
69
|
+
return "left"
|
|
70
|
+
return "stay"
|
|
71
|
+
|
|
72
|
+
@staticmethod
|
|
73
|
+
def _obs(state: Mapping[str, Any]) -> dict:
|
|
74
|
+
return {
|
|
75
|
+
"pos": state["pos"],
|
|
76
|
+
"target": state["target"],
|
|
77
|
+
"remaining": state["max_steps"] - state["steps"],
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
class GuessNumberEnv:
|
|
82
|
+
"""Binary-search style: guess ``secret`` in [low, high] with higher/lower hints.
|
|
83
|
+
|
|
84
|
+
obs = {low, high, last, hint, remaining}. Reward 1.0 on the correct guess.
|
|
85
|
+
Optimal policy = guess the midpoint. Action = the integer guess (as str).
|
|
86
|
+
"""
|
|
87
|
+
|
|
88
|
+
actions: tuple[str, ...] = () # any int in range; reference uses midpoint
|
|
89
|
+
|
|
90
|
+
def reset(self, spec: Mapping[str, Any]) -> tuple[dict, dict]:
|
|
91
|
+
low, high = int(spec.get("low", 1)), int(spec.get("high", 100))
|
|
92
|
+
state = {
|
|
93
|
+
"low": low, "high": high, "secret": int(spec.get("secret", (low + high) // 3)),
|
|
94
|
+
"last": None, "hint": "go", "steps": 0,
|
|
95
|
+
"max_steps": int(spec.get("max_steps", 12)),
|
|
96
|
+
}
|
|
97
|
+
return state, self._obs(state)
|
|
98
|
+
|
|
99
|
+
def step(self, state: dict, action: str) -> tuple[dict, dict, float, bool, dict]:
|
|
100
|
+
state = dict(state)
|
|
101
|
+
try:
|
|
102
|
+
guess = int(action)
|
|
103
|
+
except (TypeError, ValueError):
|
|
104
|
+
guess = state["low"]
|
|
105
|
+
state["steps"] += 1
|
|
106
|
+
state["last"] = guess
|
|
107
|
+
if guess == state["secret"]:
|
|
108
|
+
state["hint"] = "correct"
|
|
109
|
+
return state, self._obs(state), 1.0, True, {"reached": True}
|
|
110
|
+
if guess < state["secret"]:
|
|
111
|
+
state["low"] = guess + 1
|
|
112
|
+
state["hint"] = "higher"
|
|
113
|
+
else:
|
|
114
|
+
state["high"] = guess - 1
|
|
115
|
+
state["hint"] = "lower"
|
|
116
|
+
done = state["steps"] >= state["max_steps"]
|
|
117
|
+
return state, self._obs(state), 0.0, done, {"reached": False}
|
|
118
|
+
|
|
119
|
+
def optimal_action(self, obs: Mapping[str, Any]) -> str:
|
|
120
|
+
return str((int(obs["low"]) + int(obs["high"])) // 2)
|
|
121
|
+
|
|
122
|
+
@staticmethod
|
|
123
|
+
def _obs(state: Mapping[str, Any]) -> dict:
|
|
124
|
+
return {
|
|
125
|
+
"low": state["low"], "high": state["high"], "last": state["last"],
|
|
126
|
+
"hint": state["hint"], "remaining": state["max_steps"] - state["steps"],
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
ENVIRONMENTS: dict[str, Callable[[], Environment]] = {
|
|
131
|
+
"reach_target": lambda: ReachTargetEnv(),
|
|
132
|
+
"guess_number": lambda: GuessNumberEnv(),
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
class PullError(ValueError):
|
|
137
|
+
"""Raised for an unknown env kind or malformed pull task."""
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def resolve_policy(agent: Any, env: Environment) -> Callable[[Mapping[str, Any]], str]:
|
|
141
|
+
"""Resolve a policy from a callable or a spec dict (``reference`` / ``noop``)."""
|
|
142
|
+
|
|
143
|
+
if callable(agent):
|
|
144
|
+
return agent
|
|
145
|
+
spec = agent if isinstance(agent, Mapping) else {}
|
|
146
|
+
kind = str(spec.get("type", "reference"))
|
|
147
|
+
if kind == "reference":
|
|
148
|
+
return env.optimal_action
|
|
149
|
+
if kind == "noop":
|
|
150
|
+
first = env.actions[0] if env.actions else "0"
|
|
151
|
+
return lambda _obs: first
|
|
152
|
+
raise PullError(f"unknown pull policy {kind!r}; expected callable / reference / noop")
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def run_pull(task: Mapping[str, Any], agent: Any) -> dict[str, Any]:
|
|
156
|
+
"""Run one agent-driven episode over a simulated environment.
|
|
157
|
+
|
|
158
|
+
Returns ``{"result", "raw"}`` (unified Result). The scalar is the cumulative
|
|
159
|
+
reward; ``pass_fail`` records goal-reached; ``raw`` records the trajectory
|
|
160
|
+
length + terminal info.
|
|
161
|
+
"""
|
|
162
|
+
|
|
163
|
+
env_spec = task.get("env") or {}
|
|
164
|
+
kind = str(env_spec.get("kind") or "")
|
|
165
|
+
if kind not in ENVIRONMENTS:
|
|
166
|
+
return {
|
|
167
|
+
"result": {"scalar": 0.0, "components": {}, "pass_fail": {},
|
|
168
|
+
"explanation": f"unknown env kind {kind!r}"},
|
|
169
|
+
"raw": {"control": "pull", "infra_error": True, "env_kind": kind},
|
|
170
|
+
}
|
|
171
|
+
env = ENVIRONMENTS[kind]()
|
|
172
|
+
try:
|
|
173
|
+
policy = resolve_policy(agent, env)
|
|
174
|
+
except PullError as exc:
|
|
175
|
+
return {
|
|
176
|
+
"result": {"scalar": 0.0, "components": {}, "pass_fail": {},
|
|
177
|
+
"explanation": str(exc)},
|
|
178
|
+
"raw": {"control": "pull", "infra_error": True},
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
state, obs = env.reset(env_spec.get("spec") or {})
|
|
182
|
+
total = 0.0
|
|
183
|
+
reached = False
|
|
184
|
+
steps = 0
|
|
185
|
+
hard_cap = int((env_spec.get("spec") or {}).get("max_steps", 50)) + 5
|
|
186
|
+
while steps < hard_cap:
|
|
187
|
+
try:
|
|
188
|
+
action = policy(obs)
|
|
189
|
+
except Exception as exc: # a misbehaving policy fails the episode, not the lane
|
|
190
|
+
return {
|
|
191
|
+
"result": {"scalar": round(total, 6), "components": {"reward": total},
|
|
192
|
+
"pass_fail": {"goal_reached": False},
|
|
193
|
+
"explanation": f"policy raised: {exc}"},
|
|
194
|
+
"raw": {"control": "pull", "env_kind": kind, "steps": steps, "policy_error": True},
|
|
195
|
+
}
|
|
196
|
+
state, obs, reward, done, info = env.step(state, str(action))
|
|
197
|
+
total += float(reward)
|
|
198
|
+
steps += 1
|
|
199
|
+
if info.get("reached"):
|
|
200
|
+
reached = True
|
|
201
|
+
if done:
|
|
202
|
+
break
|
|
203
|
+
|
|
204
|
+
return {
|
|
205
|
+
"result": {
|
|
206
|
+
"scalar": round(total, 6),
|
|
207
|
+
"components": {"reward": round(total, 6), "steps": float(steps)},
|
|
208
|
+
"pass_fail": {"goal_reached": reached},
|
|
209
|
+
"explanation": f"{'reached' if reached else 'did not reach'} goal in {steps} steps",
|
|
210
|
+
},
|
|
211
|
+
"raw": {"control": "pull", "env_kind": kind, "steps": steps, "reached": reached},
|
|
212
|
+
}
|
fi/alk/bench/_voice.py
ADDED
|
@@ -0,0 +1,147 @@
|
|
|
1
|
+
"""Voice modality — deterministic voice-episode verifier (deep contract + simulated).
|
|
2
|
+
|
|
3
|
+
Voice is the modality that stress-tests the harness: the environment is an active
|
|
4
|
+
caller and the verifier is *temporal*, not an exit code. This module scores a
|
|
5
|
+
voice **episode transcript** (interleaved caller + agent turns with millisecond
|
|
6
|
+
timing) on the dimensions a real voice benchmark cares about:
|
|
7
|
+
|
|
8
|
+
* **latency** — the agent answers within the budget after the caller stops;
|
|
9
|
+
* **turn-taking** — no harmful overlap (both speaking at once) outside a
|
|
10
|
+
legitimate barge-in;
|
|
11
|
+
* **barge-in handling** — when the caller interrupts mid-agent-turn, the agent
|
|
12
|
+
yields promptly;
|
|
13
|
+
* **task content** — the agent's words cover the required content.
|
|
14
|
+
|
|
15
|
+
This is the **simulated / deep-contract** tier: it scores a transcript produced
|
|
16
|
+
by a deterministic simulated caller, so it is fully credential-free and
|
|
17
|
+
gate-verifiable. The same verifier consumes a transcript captured from a *live*
|
|
18
|
+
audio/SIP/WebRTC call + ASR — that live capture (and real WER) is deferred to
|
|
19
|
+
owner infra; it plugs in here unchanged by producing the same transcript shape.
|
|
20
|
+
|
|
21
|
+
A transcript is a list of turns::
|
|
22
|
+
|
|
23
|
+
{"speaker": "caller"|"agent", "start_ms": int, "end_ms": int,
|
|
24
|
+
"text": str, "interrupt": bool (optional, caller turns only)}
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
from __future__ import annotations
|
|
28
|
+
|
|
29
|
+
from typing import Any, Mapping, Sequence
|
|
30
|
+
|
|
31
|
+
_DEFAULT_MAX_LATENCY_MS = 1200
|
|
32
|
+
_BARGE_IN_YIELD_MS = 600 # the agent must stop within this of a barge-in to "yield"
|
|
33
|
+
_PASS_FLOOR = 0.75 # each sub-score must meet this for a pass
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def _norm_turns(dialogue: Sequence[Mapping[str, Any]]) -> list[dict[str, Any]]:
|
|
37
|
+
turns = []
|
|
38
|
+
for t in dialogue:
|
|
39
|
+
turns.append({
|
|
40
|
+
"speaker": str(t.get("speaker")),
|
|
41
|
+
"start_ms": int(t.get("start_ms", 0)),
|
|
42
|
+
"end_ms": int(t.get("end_ms", 0)),
|
|
43
|
+
"text": str(t.get("text") or ""),
|
|
44
|
+
"interrupt": bool(t.get("interrupt", False)),
|
|
45
|
+
})
|
|
46
|
+
return turns
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _latency_score(turns: list[dict[str, Any]], max_latency_ms: int) -> float:
|
|
50
|
+
gaps_ok, gaps = 0, 0
|
|
51
|
+
for i, t in enumerate(turns):
|
|
52
|
+
if t["speaker"] != "agent":
|
|
53
|
+
continue
|
|
54
|
+
prev = next((turns[j] for j in range(i - 1, -1, -1)
|
|
55
|
+
if turns[j]["speaker"] == "caller"), None)
|
|
56
|
+
if prev is None:
|
|
57
|
+
continue
|
|
58
|
+
gap = t["start_ms"] - prev["end_ms"]
|
|
59
|
+
gaps += 1
|
|
60
|
+
if 0 <= gap <= max_latency_ms:
|
|
61
|
+
gaps_ok += 1
|
|
62
|
+
return gaps_ok / gaps if gaps else 1.0
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def _overlap_and_bargein(turns: list[dict[str, Any]]) -> tuple[float, float]:
|
|
66
|
+
"""Turn-taking (no harmful overlap) + barge-in handling scores."""
|
|
67
|
+
|
|
68
|
+
agent_turns = [t for t in turns if t["speaker"] == "agent"]
|
|
69
|
+
harmful, considered = 0, 0
|
|
70
|
+
bargein_handled, bargein_total = 0, 0
|
|
71
|
+
callers = [t for t in turns if t["speaker"] == "caller"]
|
|
72
|
+
for a in agent_turns:
|
|
73
|
+
considered += 1
|
|
74
|
+
overlapping_callers = [
|
|
75
|
+
c for c in callers
|
|
76
|
+
if c["start_ms"] < a["end_ms"] and c["end_ms"] > a["start_ms"]
|
|
77
|
+
]
|
|
78
|
+
legit_bargein = False
|
|
79
|
+
for c in overlapping_callers:
|
|
80
|
+
if c["interrupt"]:
|
|
81
|
+
bargein_total += 1
|
|
82
|
+
legit_bargein = True
|
|
83
|
+
# the agent must yield: its turn ends within the window after the
|
|
84
|
+
# interrupt begins.
|
|
85
|
+
if a["end_ms"] - c["start_ms"] <= _BARGE_IN_YIELD_MS:
|
|
86
|
+
bargein_handled += 1
|
|
87
|
+
if overlapping_callers and not legit_bargein:
|
|
88
|
+
harmful += 1 # both speaking at once with no barge-in to excuse it
|
|
89
|
+
turn_taking = 1.0 - (harmful / considered) if considered else 1.0
|
|
90
|
+
bargein = bargein_handled / bargein_total if bargein_total else 1.0
|
|
91
|
+
return turn_taking, bargein
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _content_score(turns: list[dict[str, Any]], required: Sequence[str]) -> float:
|
|
95
|
+
if not required:
|
|
96
|
+
return 1.0
|
|
97
|
+
agent_text = " ".join(t["text"] for t in turns if t["speaker"] == "agent").lower()
|
|
98
|
+
hit = sum(1 for kw in required if str(kw).lower() in agent_text)
|
|
99
|
+
return hit / len(required)
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def score_voice_episode(
|
|
103
|
+
dialogue: Sequence[Mapping[str, Any]],
|
|
104
|
+
*,
|
|
105
|
+
budgets: Mapping[str, Any] | None = None,
|
|
106
|
+
required_content: Sequence[str] | None = None,
|
|
107
|
+
) -> dict[str, Any]:
|
|
108
|
+
"""Score a voice episode transcript; return ``{"result", "raw"}`` (unified Result).
|
|
109
|
+
|
|
110
|
+
The scalar is the mean of the four sub-scores; the verdict (in pass_fail) is a
|
|
111
|
+
pass only if EVERY sub-score meets the floor — a single bad dimension (e.g. the
|
|
112
|
+
agent talks over the caller) fails the episode.
|
|
113
|
+
"""
|
|
114
|
+
|
|
115
|
+
budgets = budgets or {}
|
|
116
|
+
required_content = required_content or []
|
|
117
|
+
if not dialogue:
|
|
118
|
+
return {
|
|
119
|
+
"result": {"scalar": 0.0, "components": {}, "pass_fail": {"voice": False},
|
|
120
|
+
"explanation": "empty transcript"},
|
|
121
|
+
"raw": {"modality": "voice"},
|
|
122
|
+
}
|
|
123
|
+
turns = _norm_turns(dialogue)
|
|
124
|
+
max_latency = int(budgets.get("max_latency_ms", _DEFAULT_MAX_LATENCY_MS))
|
|
125
|
+
latency = _latency_score(turns, max_latency)
|
|
126
|
+
turn_taking, bargein = _overlap_and_bargein(turns)
|
|
127
|
+
content = _content_score(turns, required_content)
|
|
128
|
+
sub = {
|
|
129
|
+
"latency": round(latency, 6),
|
|
130
|
+
"turn_taking": round(turn_taking, 6),
|
|
131
|
+
"barge_in": round(bargein, 6),
|
|
132
|
+
"content": round(content, 6),
|
|
133
|
+
}
|
|
134
|
+
scalar = round(sum(sub.values()) / len(sub), 6)
|
|
135
|
+
floors_met = all(v >= _PASS_FLOOR for v in sub.values())
|
|
136
|
+
return {
|
|
137
|
+
"result": {
|
|
138
|
+
"scalar": scalar,
|
|
139
|
+
"components": sub,
|
|
140
|
+
"pass_fail": {"voice": floors_met, **{f"{k}_floor": (v >= _PASS_FLOOR)
|
|
141
|
+
for k, v in sub.items()}},
|
|
142
|
+
"explanation": ("all voice dimensions met the floor" if floors_met
|
|
143
|
+
else "a voice dimension fell below the floor"),
|
|
144
|
+
},
|
|
145
|
+
"raw": {"modality": "voice", "turns": len(turns), "max_latency_ms": max_latency,
|
|
146
|
+
"floors_met": floors_met},
|
|
147
|
+
}
|