agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,213 @@
|
|
|
1
|
+
"""Code-tests verifier — run held-out tests against candidate code in isolation.
|
|
2
|
+
|
|
3
|
+
This is the coding-modality verifier for the bench harness. It implements the
|
|
4
|
+
trustworthiness rule every serious coding benchmark uses: **the oracle is held
|
|
5
|
+
out** — the held-out checks live in a separate file the candidate code never
|
|
6
|
+
imports or sees, and they are executed by a harness-written runner, not by the
|
|
7
|
+
candidate.
|
|
8
|
+
|
|
9
|
+
Sandboxes:
|
|
10
|
+
* ``subprocess`` (default) — a fresh interpreter in a throwaway tempdir, with a
|
|
11
|
+
scrubbed environment (no harness secrets) and a hard wall-clock timeout. This
|
|
12
|
+
is the sandbox used by the credential-free release gate, which only ever runs
|
|
13
|
+
trusted, shipped reference code. It is **not** a security boundary against
|
|
14
|
+
deliberately hostile code (no real filesystem/network isolation); for
|
|
15
|
+
untrusted agent output use the Docker lane (bench step 15E).
|
|
16
|
+
* ``docker`` — per-task container isolation with a no-network default
|
|
17
|
+
(bench step 15E).
|
|
18
|
+
|
|
19
|
+
The convention for a checks file: it defines one or more ``check_*`` callables
|
|
20
|
+
that import the candidate module (``import solution``) and ``assert`` the
|
|
21
|
+
expected behaviour. The harness discovers them, runs each, and reports per-check
|
|
22
|
+
pass/fail — so a candidate that no-ops, prints a fake "success" message, returns
|
|
23
|
+
wrong answers, or fails to define the entrypoint is failed deterministically.
|
|
24
|
+
|
|
25
|
+
THREAT-MODEL NOTE: the runner and the candidate share one process, so the
|
|
26
|
+
deterministic-failure guarantee covers *accidental* gaming, not an *adversarial*
|
|
27
|
+
candidate. A candidate that knows this runner's protocol could, during its import
|
|
28
|
+
body (which runs before ``check_*``), print a forged ``{"results": ...}`` line and
|
|
29
|
+
exit 0, or read the checks file to reflect expected values. Hardening that
|
|
30
|
+
(process/UID separation of the oracle + an authenticated out-of-band verdict
|
|
31
|
+
channel — the inject-tests-after-the-agent-finishes topology) is tracked separate
|
|
32
|
+
work; until then do not treat a passing score from an untrusted adversarial
|
|
33
|
+
candidate as authoritative. The release gate is unaffected: it runs only trusted
|
|
34
|
+
shipped reference code.
|
|
35
|
+
"""
|
|
36
|
+
|
|
37
|
+
from __future__ import annotations
|
|
38
|
+
|
|
39
|
+
import json
|
|
40
|
+
import subprocess
|
|
41
|
+
import sys
|
|
42
|
+
import tempfile
|
|
43
|
+
from pathlib import Path
|
|
44
|
+
from typing import Any
|
|
45
|
+
|
|
46
|
+
from ..live._runner import scrubbed_lane_env
|
|
47
|
+
|
|
48
|
+
#: Entry module the candidate code is written to (checks ``import`` this name).
|
|
49
|
+
ENTRY_MODULE = "solution"
|
|
50
|
+
_CHECKS_MODULE = "bench_checks"
|
|
51
|
+
_DEFAULT_TIMEOUT_S = 10.0
|
|
52
|
+
|
|
53
|
+
SUPPORTED_LANGUAGES = ("python",)
|
|
54
|
+
|
|
55
|
+
# The harness-written runner: discovers ``check_*`` callables in the checks
|
|
56
|
+
# module, runs each, and emits a single JSON line of per-check results to stdout.
|
|
57
|
+
# The candidate (``solution.py``) is imported only by the checks module — never
|
|
58
|
+
# by this runner directly — so the oracle stays out of the candidate's reach.
|
|
59
|
+
_PYTHON_RUNNER = """\
|
|
60
|
+
import importlib, json, sys, traceback
|
|
61
|
+
results = {}
|
|
62
|
+
fatal = None
|
|
63
|
+
try:
|
|
64
|
+
checks = importlib.import_module("%(checks)s")
|
|
65
|
+
except Exception:
|
|
66
|
+
fatal = "checks_import_failed: " + traceback.format_exc(limit=2).strip().replace(chr(10), " | ")
|
|
67
|
+
print(json.dumps({"results": {}, "fatal": fatal}))
|
|
68
|
+
sys.exit(1)
|
|
69
|
+
names = sorted(n for n in dir(checks) if n.startswith("check_") and callable(getattr(checks, n)))
|
|
70
|
+
if not names:
|
|
71
|
+
print(json.dumps({"results": {}, "fatal": "no check_* callables found"}))
|
|
72
|
+
sys.exit(1)
|
|
73
|
+
for name in names:
|
|
74
|
+
try:
|
|
75
|
+
getattr(checks, name)()
|
|
76
|
+
results[name] = True
|
|
77
|
+
except Exception:
|
|
78
|
+
results[name] = False
|
|
79
|
+
print(json.dumps({"results": results, "fatal": None}))
|
|
80
|
+
sys.exit(0 if results and all(results.values()) else 1)
|
|
81
|
+
"""
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def _tail(text: str, limit: int = 2000) -> str:
|
|
85
|
+
text = text or ""
|
|
86
|
+
return text[-limit:]
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def _empty_result(explanation: str, raw: dict[str, Any]) -> dict[str, Any]:
|
|
90
|
+
return {
|
|
91
|
+
"result": {
|
|
92
|
+
"scalar": 0.0,
|
|
93
|
+
"components": {"checks_passed": 0.0, "checks_total": 0.0},
|
|
94
|
+
"pass_fail": {},
|
|
95
|
+
"explanation": explanation,
|
|
96
|
+
},
|
|
97
|
+
"raw": raw,
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def run_code_tests(
|
|
102
|
+
candidate_code: str,
|
|
103
|
+
checks_code: str,
|
|
104
|
+
*,
|
|
105
|
+
language: str = "python",
|
|
106
|
+
timeout_s: float = _DEFAULT_TIMEOUT_S,
|
|
107
|
+
sandbox: str = "subprocess",
|
|
108
|
+
) -> dict[str, Any]:
|
|
109
|
+
"""Run ``checks_code`` (the held-out oracle) against ``candidate_code``.
|
|
110
|
+
|
|
111
|
+
Returns ``{"result": <unified Result>, "raw": <execution evidence>}``. The
|
|
112
|
+
unified ``Result`` carries a scalar (fraction of checks passed), components
|
|
113
|
+
(passed/total), per-check ``pass_fail`` booleans, and an explanation.
|
|
114
|
+
"""
|
|
115
|
+
|
|
116
|
+
if language not in SUPPORTED_LANGUAGES:
|
|
117
|
+
return _empty_result(
|
|
118
|
+
f"unsupported language {language!r}; supported: {SUPPORTED_LANGUAGES}",
|
|
119
|
+
{"sandbox": sandbox, "language": language, "infra_error": True},
|
|
120
|
+
)
|
|
121
|
+
if sandbox == "docker":
|
|
122
|
+
# The Docker lane is opt-in; never silently fall back to a weaker sandbox
|
|
123
|
+
# (that would mislabel isolation).
|
|
124
|
+
from ._docker import run_code_tests_docker # local import: optional lane
|
|
125
|
+
|
|
126
|
+
return run_code_tests_docker(
|
|
127
|
+
candidate_code, checks_code, language=language, timeout_s=timeout_s
|
|
128
|
+
)
|
|
129
|
+
if sandbox != "subprocess":
|
|
130
|
+
return _empty_result(
|
|
131
|
+
f"unknown sandbox {sandbox!r}; expected 'subprocess' or 'docker'",
|
|
132
|
+
{"sandbox": sandbox, "language": language, "infra_error": True},
|
|
133
|
+
)
|
|
134
|
+
|
|
135
|
+
return _run_subprocess(candidate_code, checks_code, timeout_s=timeout_s)
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def _run_subprocess(
|
|
139
|
+
candidate_code: str, checks_code: str, *, timeout_s: float
|
|
140
|
+
) -> dict[str, Any]:
|
|
141
|
+
with tempfile.TemporaryDirectory(prefix="agent-learn-bench-") as tmp:
|
|
142
|
+
root = Path(tmp)
|
|
143
|
+
(root / f"{ENTRY_MODULE}.py").write_text(candidate_code, encoding="utf-8")
|
|
144
|
+
(root / f"{_CHECKS_MODULE}.py").write_text(checks_code, encoding="utf-8")
|
|
145
|
+
(root / "_runner.py").write_text(
|
|
146
|
+
_PYTHON_RUNNER % {"checks": _CHECKS_MODULE}, encoding="utf-8"
|
|
147
|
+
)
|
|
148
|
+
|
|
149
|
+
raw: dict[str, Any] = {
|
|
150
|
+
"sandbox": "subprocess",
|
|
151
|
+
"language": "python",
|
|
152
|
+
"timed_out": False,
|
|
153
|
+
"exit_code": None,
|
|
154
|
+
}
|
|
155
|
+
try:
|
|
156
|
+
proc = subprocess.run(
|
|
157
|
+
[sys.executable, "_runner.py"],
|
|
158
|
+
cwd=str(root),
|
|
159
|
+
env=scrubbed_lane_env(()), # no harness secrets cross into the run
|
|
160
|
+
capture_output=True,
|
|
161
|
+
text=True,
|
|
162
|
+
timeout=timeout_s,
|
|
163
|
+
)
|
|
164
|
+
except subprocess.TimeoutExpired as exc:
|
|
165
|
+
raw["timed_out"] = True
|
|
166
|
+
raw["stdout_tail"] = _tail(exc.stdout if isinstance(exc.stdout, str) else "")
|
|
167
|
+
raw["stderr_tail"] = _tail(exc.stderr if isinstance(exc.stderr, str) else "")
|
|
168
|
+
return _empty_result(f"timed out after {timeout_s}s", raw)
|
|
169
|
+
|
|
170
|
+
raw["exit_code"] = proc.returncode
|
|
171
|
+
raw["stdout_tail"] = _tail(proc.stdout)
|
|
172
|
+
raw["stderr_tail"] = _tail(proc.stderr)
|
|
173
|
+
|
|
174
|
+
parsed = _parse_runner_stdout(proc.stdout)
|
|
175
|
+
if parsed is None:
|
|
176
|
+
return _empty_result(
|
|
177
|
+
f"runner produced no parseable result (exit {proc.returncode})", raw
|
|
178
|
+
)
|
|
179
|
+
fatal = parsed.get("fatal")
|
|
180
|
+
results = {str(k): bool(v) for k, v in (parsed.get("results") or {}).items()}
|
|
181
|
+
if fatal:
|
|
182
|
+
return _empty_result(str(fatal), raw)
|
|
183
|
+
if not results:
|
|
184
|
+
return _empty_result("no checks executed", raw)
|
|
185
|
+
|
|
186
|
+
total = len(results)
|
|
187
|
+
passed = sum(1 for v in results.values() if v)
|
|
188
|
+
return {
|
|
189
|
+
"result": {
|
|
190
|
+
"scalar": round(passed / total, 6),
|
|
191
|
+
"components": {
|
|
192
|
+
"checks_passed": float(passed),
|
|
193
|
+
"checks_total": float(total),
|
|
194
|
+
},
|
|
195
|
+
"pass_fail": results,
|
|
196
|
+
"explanation": f"{passed}/{total} checks passed",
|
|
197
|
+
},
|
|
198
|
+
"raw": raw,
|
|
199
|
+
}
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
def _parse_runner_stdout(stdout: str) -> dict[str, Any] | None:
|
|
203
|
+
# The runner prints exactly one JSON line; tolerate trailing candidate prints
|
|
204
|
+
# by scanning for the last decodable JSON object.
|
|
205
|
+
for line in reversed((stdout or "").splitlines()):
|
|
206
|
+
line = line.strip()
|
|
207
|
+
if not line.startswith("{"):
|
|
208
|
+
continue
|
|
209
|
+
try:
|
|
210
|
+
return json.loads(line)
|
|
211
|
+
except json.JSONDecodeError:
|
|
212
|
+
continue
|
|
213
|
+
return None
|
fi/alk/bench/_coding.py
ADDED
|
@@ -0,0 +1,215 @@
|
|
|
1
|
+
"""Coding-modality bench suite + the artifact-in runner.
|
|
2
|
+
|
|
3
|
+
A coding suite is a bench-native shape (``agent-learning.bench-suite.v1``): each
|
|
4
|
+
task carries an ``instruction``, a held-out ``checks`` oracle (executed against
|
|
5
|
+
the candidate, never imported by it), a ``reference_solution`` (the gold, used by
|
|
6
|
+
the release gate to prove the verifier accepts a correct answer), and optional
|
|
7
|
+
``guards``. This is deliberately distinct from the objective-anchored task
|
|
8
|
+
dataset: coding's verdict is "do the held-out tests pass", not a weighted-metric
|
|
9
|
+
mean. The two unify at the Result level, not the suite level — exactly the shape
|
|
10
|
+
the prior-art survey found (task specs are modality-specific; the Task<->Verifier
|
|
11
|
+
coupling and the unified Result are the invariant).
|
|
12
|
+
|
|
13
|
+
``artifact_in`` control mode scores a *submitted artifact* (candidate code) with
|
|
14
|
+
no live agent — the analogue of patch-scoring harnesses. The agent that produced
|
|
15
|
+
the artifact is out of scope here; only the held-out oracle decides the verdict.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import json
|
|
21
|
+
from pathlib import Path
|
|
22
|
+
from typing import Any, Mapping
|
|
23
|
+
|
|
24
|
+
from ._codeexec import run_code_tests
|
|
25
|
+
from ._grader import GRADING_COMMAND, run_command_graded
|
|
26
|
+
|
|
27
|
+
BENCH_SUITE_KIND = "agent-learning.bench-suite.v1"
|
|
28
|
+
|
|
29
|
+
# A task is graded one of two ways:
|
|
30
|
+
# * "checks" (default, convenience tier): held-out check_* functions import the
|
|
31
|
+
# candidate in-process. Trusted / accidental-gaming only.
|
|
32
|
+
# * "command" (hardened tier): the candidate produces files/output, a held-out
|
|
33
|
+
# grader runs AFTER and emits the verdict via exit code + reward file. Robust
|
|
34
|
+
# against forge + oracle-read; multi-language. See _grader.py.
|
|
35
|
+
_CHECKS_FIELDS = ("id", "instruction", "checks", "reference_solution")
|
|
36
|
+
_COMMAND_FIELDS = ("id", "instruction", "grader_cmd", "grader_files", "reference_files")
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class CodingSuiteError(ValueError):
|
|
40
|
+
"""Raised for a malformed coding bench suite."""
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def is_bench_suite(obj: Any) -> bool:
|
|
44
|
+
return isinstance(obj, Mapping) and obj.get("kind") == BENCH_SUITE_KIND
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def _task_grading(task: Mapping[str, Any]) -> str:
|
|
48
|
+
"""The grading mode of a task: 'command' (hardened) or 'checks' (convenience)."""
|
|
49
|
+
|
|
50
|
+
mode = task.get("grading")
|
|
51
|
+
if mode in (GRADING_COMMAND, "checks"):
|
|
52
|
+
return str(mode)
|
|
53
|
+
# infer: a grader_cmd ⇒ command-graded; otherwise the legacy checks tier.
|
|
54
|
+
return GRADING_COMMAND if task.get("grader_cmd") else "checks"
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def load_coding_suite(obj: Mapping[str, Any] | str | Path) -> dict[str, Any]:
|
|
58
|
+
"""Load + validate a coding bench suite (from a path or an in-memory mapping)."""
|
|
59
|
+
|
|
60
|
+
if isinstance(obj, (str, Path)):
|
|
61
|
+
data: Mapping[str, Any] = json.loads(Path(obj).expanduser().read_text("utf-8"))
|
|
62
|
+
else:
|
|
63
|
+
data = obj
|
|
64
|
+
if not is_bench_suite(data):
|
|
65
|
+
raise CodingSuiteError(f"not a {BENCH_SUITE_KIND} suite")
|
|
66
|
+
tasks = data.get("tasks")
|
|
67
|
+
if not isinstance(tasks, list) or not tasks:
|
|
68
|
+
raise CodingSuiteError("coding suite has no tasks")
|
|
69
|
+
seen: set[str] = set()
|
|
70
|
+
for i, task in enumerate(tasks):
|
|
71
|
+
if not isinstance(task, Mapping):
|
|
72
|
+
raise CodingSuiteError(f"task #{i} is not an object")
|
|
73
|
+
required = _COMMAND_FIELDS if _task_grading(task) == GRADING_COMMAND else _CHECKS_FIELDS
|
|
74
|
+
for field in required:
|
|
75
|
+
if not task.get(field):
|
|
76
|
+
raise CodingSuiteError(
|
|
77
|
+
f"task #{i} ({_task_grading(task)}-graded) missing required field {field!r}"
|
|
78
|
+
)
|
|
79
|
+
tid = str(task["id"])
|
|
80
|
+
if tid in seen:
|
|
81
|
+
raise CodingSuiteError(f"duplicate task id {tid!r}")
|
|
82
|
+
seen.add(tid)
|
|
83
|
+
# Every task must declare at least one guard against reward hacking; the
|
|
84
|
+
# held-out oracle is the primary defence, but the suite must say so.
|
|
85
|
+
guards = task.get("guards") or {}
|
|
86
|
+
if int(guards.get("min_guard_count", 0)) < 1:
|
|
87
|
+
raise CodingSuiteError(
|
|
88
|
+
f"task {tid!r} must declare guards.min_guard_count >= 1 "
|
|
89
|
+
"(the held-out-oracle anti-gaming contract)"
|
|
90
|
+
)
|
|
91
|
+
return dict(data)
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _coding_row(
|
|
95
|
+
task: Mapping[str, Any],
|
|
96
|
+
verdict_obj: Mapping[str, Any],
|
|
97
|
+
*,
|
|
98
|
+
evidence_class: str,
|
|
99
|
+
sandbox: str,
|
|
100
|
+
) -> dict[str, Any]:
|
|
101
|
+
result = dict(verdict_obj["result"])
|
|
102
|
+
scalar = result.get("scalar")
|
|
103
|
+
# All-or-nothing: a coding task is resolved only if EVERY held-out check passes.
|
|
104
|
+
verdict = "pass" if scalar is not None and float(scalar) >= 1.0 else "fail"
|
|
105
|
+
return {
|
|
106
|
+
"task_id": str(task["id"]),
|
|
107
|
+
"modality": "coding",
|
|
108
|
+
"world_kind": "code_exec",
|
|
109
|
+
"control_mode": "artifact_in",
|
|
110
|
+
"result": result,
|
|
111
|
+
"verdict": verdict,
|
|
112
|
+
# The candidate code really executed; honest execution_class is executable.
|
|
113
|
+
"execution_class": "executable",
|
|
114
|
+
"evidence_class": evidence_class,
|
|
115
|
+
# executable + any evidence class is never an overclaim (it really ran).
|
|
116
|
+
"overclaim": False,
|
|
117
|
+
"sandbox": sandbox,
|
|
118
|
+
"raw": verdict_obj.get("raw", {}),
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def _void_row(task: Mapping[str, Any], reason: str, *, evidence_class: str) -> dict[str, Any]:
|
|
123
|
+
return {
|
|
124
|
+
"task_id": str(task["id"]),
|
|
125
|
+
"modality": "coding",
|
|
126
|
+
"world_kind": "code_exec",
|
|
127
|
+
"control_mode": "artifact_in",
|
|
128
|
+
"result": {
|
|
129
|
+
"scalar": None,
|
|
130
|
+
"components": {},
|
|
131
|
+
"pass_fail": {},
|
|
132
|
+
"explanation": reason,
|
|
133
|
+
},
|
|
134
|
+
"verdict": "void",
|
|
135
|
+
"execution_class": "executable",
|
|
136
|
+
"evidence_class": evidence_class,
|
|
137
|
+
"overclaim": False,
|
|
138
|
+
"error": reason,
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def run_coding_artifact_in(
|
|
143
|
+
suite: Mapping[str, Any],
|
|
144
|
+
submission: Mapping[str, Any],
|
|
145
|
+
*,
|
|
146
|
+
sandbox: str = "subprocess",
|
|
147
|
+
evidence_class: str = "captured_fixture",
|
|
148
|
+
max_tasks: int | None = None,
|
|
149
|
+
default_timeout_s: float = 10.0,
|
|
150
|
+
) -> list[dict[str, Any]]:
|
|
151
|
+
"""Score each task's submitted artifact against its held-out oracle.
|
|
152
|
+
|
|
153
|
+
``submission`` maps ``task_id -> candidate``. For ``checks``-graded tasks the
|
|
154
|
+
candidate is a source string; for ``command``-graded (hardened) tasks it is a
|
|
155
|
+
``{path: content}`` file map. A task with no submission is recorded ``void``
|
|
156
|
+
(never silently passed); an infra failure is recorded ``void`` too. Pass the
|
|
157
|
+
:func:`reference_submission` to verify the suite itself (what the gate does).
|
|
158
|
+
"""
|
|
159
|
+
|
|
160
|
+
language = str(suite.get("language", "python"))
|
|
161
|
+
# Honesty: a Docker run executes untrusted candidate code under real
|
|
162
|
+
# isolation -> that is a genuine LIVE event, never a fixture/local class.
|
|
163
|
+
# Force at least live_lane (honor an explicit live_stressed); never downgrade.
|
|
164
|
+
if sandbox == "docker" and evidence_class not in ("live_lane", "live_stressed"):
|
|
165
|
+
evidence_class = "live_lane"
|
|
166
|
+
rows: list[dict[str, Any]] = []
|
|
167
|
+
tasks = suite["tasks"]
|
|
168
|
+
if max_tasks is not None:
|
|
169
|
+
tasks = tasks[: max(0, int(max_tasks))]
|
|
170
|
+
for task in tasks:
|
|
171
|
+
tid = str(task["id"])
|
|
172
|
+
candidate = submission.get(tid)
|
|
173
|
+
if candidate is None:
|
|
174
|
+
rows.append(_void_row(task, "no submission provided", evidence_class=evidence_class))
|
|
175
|
+
continue
|
|
176
|
+
timeout_s = float(task.get("timeout_s", default_timeout_s))
|
|
177
|
+
if _task_grading(task) == GRADING_COMMAND:
|
|
178
|
+
files = candidate if isinstance(candidate, Mapping) else {"solution": str(candidate)}
|
|
179
|
+
verdict_obj = run_command_graded(
|
|
180
|
+
task, {str(k): str(v) for k, v in files.items()},
|
|
181
|
+
sandbox=sandbox, timeout_s=timeout_s,
|
|
182
|
+
)
|
|
183
|
+
else:
|
|
184
|
+
verdict_obj = run_code_tests(
|
|
185
|
+
str(candidate), str(task["checks"]),
|
|
186
|
+
language=language, timeout_s=timeout_s, sandbox=sandbox,
|
|
187
|
+
)
|
|
188
|
+
# An infra/config failure (no Docker daemon, image pull failure, bad
|
|
189
|
+
# sandbox/language) means the lane never ran — record it as VOID, never as
|
|
190
|
+
# a real "fail". Conflating "the daemon was missing" with "the agent was
|
|
191
|
+
# wrong" would silently report a correct agent at 0%.
|
|
192
|
+
if (verdict_obj.get("raw") or {}).get("infra_error"):
|
|
193
|
+
reason = (verdict_obj.get("result") or {}).get("explanation") or "infrastructure error"
|
|
194
|
+
rows.append(_void_row(task, f"infra: {reason}", evidence_class=evidence_class))
|
|
195
|
+
continue
|
|
196
|
+
rows.append(
|
|
197
|
+
_coding_row(task, verdict_obj, evidence_class=evidence_class, sandbox=sandbox)
|
|
198
|
+
)
|
|
199
|
+
return rows
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
def reference_submission(suite: Mapping[str, Any]) -> dict[str, Any]:
|
|
203
|
+
"""The gold submission: every task id -> its reference.
|
|
204
|
+
|
|
205
|
+
``checks``-graded tasks map to a ``reference_solution`` string; ``command``-
|
|
206
|
+
graded tasks map to a ``reference_files`` ``{path: content}`` map.
|
|
207
|
+
"""
|
|
208
|
+
|
|
209
|
+
out: dict[str, Any] = {}
|
|
210
|
+
for t in suite["tasks"]:
|
|
211
|
+
if _task_grading(t) == GRADING_COMMAND:
|
|
212
|
+
out[str(t["id"])] = {str(k): str(v) for k, v in (t["reference_files"] or {}).items()}
|
|
213
|
+
else:
|
|
214
|
+
out[str(t["id"])] = str(t["reference_solution"])
|
|
215
|
+
return out
|
fi/alk/bench/_docker.py
ADDED
|
@@ -0,0 +1,237 @@
|
|
|
1
|
+
"""Docker code-exec lane — run held-out checks against untrusted candidate code
|
|
2
|
+
in a per-task, network-isolated, resource-capped, ephemeral container.
|
|
3
|
+
|
|
4
|
+
This is the harder-isolated sibling of the subprocess verifier in ``_codeexec``.
|
|
5
|
+
For **untrusted agent output** it adds real OS-level isolation the subprocess lane
|
|
6
|
+
cannot: no network, no host writes, dropped capabilities + no-new-privileges, a
|
|
7
|
+
nosuid tmpfs, capped CPU/memory/PIDs, killed + removed after the run.
|
|
8
|
+
|
|
9
|
+
Honesty: a Docker run of untrusted candidate code is a genuine **live** event,
|
|
10
|
+
so :func:`fi.alk.bench._coding.run_coding_artifact_in` stamps these rows
|
|
11
|
+
``evidence_class=live_lane`` (never ``captured_fixture``) — see that module.
|
|
12
|
+
|
|
13
|
+
This lane is **opt-in** (``sandbox="docker"``) and is NEVER a release-gate
|
|
14
|
+
prerequisite: the credential-free ``bench_contract_readiness`` gate runs the
|
|
15
|
+
subprocess lane on trusted shipped code so it works anywhere with no Docker. The
|
|
16
|
+
import is lazy (only resolved when ``sandbox="docker"`` is requested), so the kit
|
|
17
|
+
imports fine on a machine with no Docker.
|
|
18
|
+
|
|
19
|
+
KNOWN LIMITATION (oracle hold-out): hold-out here is only *structural* — the
|
|
20
|
+
checks file is not part of the candidate's source, but it is materialised into the
|
|
21
|
+
same container the candidate runs in, and the candidate's module body executes
|
|
22
|
+
(at import) before the ``check_*`` functions. A deliberately adversarial candidate
|
|
23
|
+
could therefore read the checks file at runtime and reflect the expected values,
|
|
24
|
+
or print a forged result line. This lane defends against *accidental* gaming
|
|
25
|
+
(no-op / fake-success / wrong answer all fail) and gives strong OS isolation, but
|
|
26
|
+
it is NOT yet a hardened defence against a candidate that actively attacks the
|
|
27
|
+
harness protocol. Closing that requires process/UID separation of the oracle from
|
|
28
|
+
the candidate (the inject-tests-only-after-the-agent-finishes topology), tracked
|
|
29
|
+
as the live-agent-in-container step. Do not treat a passing score from an
|
|
30
|
+
untrusted, adversarial candidate as authoritative until that lands.
|
|
31
|
+
"""
|
|
32
|
+
|
|
33
|
+
from __future__ import annotations
|
|
34
|
+
|
|
35
|
+
import base64
|
|
36
|
+
import shutil
|
|
37
|
+
import subprocess
|
|
38
|
+
import uuid
|
|
39
|
+
from typing import Any
|
|
40
|
+
|
|
41
|
+
from ._codeexec import (
|
|
42
|
+
SUPPORTED_LANGUAGES,
|
|
43
|
+
_empty_result,
|
|
44
|
+
_parse_runner_stdout,
|
|
45
|
+
_tail,
|
|
46
|
+
)
|
|
47
|
+
|
|
48
|
+
# In-container bootstrap: materialise the candidate + held-out checks from base64
|
|
49
|
+
# into the writable tmpfs (no host bind-mount — works identically on macOS Docker
|
|
50
|
+
# Desktop and Linux), then run each check_* in isolation and emit one JSON line.
|
|
51
|
+
# Doubled braces are literal dict syntax preserved through ``str.format``.
|
|
52
|
+
_DOCKER_BOOTSTRAP = (
|
|
53
|
+
"import base64,importlib,json,sys,traceback\n"
|
|
54
|
+
"open('/tmp/solution.py','wb').write(base64.b64decode('{cand_b64}'))\n"
|
|
55
|
+
"open('/tmp/bench_checks.py','wb').write(base64.b64decode('{checks_b64}'))\n"
|
|
56
|
+
"sys.path.insert(0,'/tmp')\n"
|
|
57
|
+
"results={{}}\n"
|
|
58
|
+
"try:\n"
|
|
59
|
+
" checks=importlib.import_module('bench_checks')\n"
|
|
60
|
+
"except Exception:\n"
|
|
61
|
+
" print(json.dumps({{'results':{{}},'fatal':'checks_import_failed: '"
|
|
62
|
+
"+traceback.format_exc(limit=2).strip().replace(chr(10),' | ')}}));sys.exit(1)\n"
|
|
63
|
+
"names=sorted(n for n in dir(checks) if n.startswith('check_') and callable(getattr(checks,n)))\n"
|
|
64
|
+
"if not names:\n"
|
|
65
|
+
" print(json.dumps({{'results':{{}},'fatal':'no check_* callables found'}}));sys.exit(1)\n"
|
|
66
|
+
"for name in names:\n"
|
|
67
|
+
" try:\n"
|
|
68
|
+
" getattr(checks,name)();results[name]=True\n"
|
|
69
|
+
" except Exception:\n"
|
|
70
|
+
" results[name]=False\n"
|
|
71
|
+
"print(json.dumps({{'results':results,'fatal':None}}))\n"
|
|
72
|
+
"sys.exit(0 if results and all(results.values()) else 1)\n"
|
|
73
|
+
)
|
|
74
|
+
|
|
75
|
+
# Default base image. Production should pin by digest (image@sha256:...) for
|
|
76
|
+
# determinism; the tag default keeps the example/proof portable.
|
|
77
|
+
DEFAULT_IMAGE = "python:3.11-slim"
|
|
78
|
+
|
|
79
|
+
_DEFAULT_MEMORY = "256m"
|
|
80
|
+
_DEFAULT_CPUS = "1.0"
|
|
81
|
+
_DEFAULT_PIDS = 128
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def docker_available() -> bool:
|
|
85
|
+
"""True if a working Docker daemon is reachable (cheap, no pull)."""
|
|
86
|
+
|
|
87
|
+
if shutil.which("docker") is None:
|
|
88
|
+
return False
|
|
89
|
+
try:
|
|
90
|
+
proc = subprocess.run(
|
|
91
|
+
["docker", "info", "--format", "{{.ServerVersion}}"],
|
|
92
|
+
capture_output=True,
|
|
93
|
+
text=True,
|
|
94
|
+
timeout=15,
|
|
95
|
+
)
|
|
96
|
+
except Exception:
|
|
97
|
+
return False
|
|
98
|
+
return proc.returncode == 0
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def _build_docker_argv(
|
|
102
|
+
name: str, image: str, memory: str, cpus: str, bootstrap: str
|
|
103
|
+
) -> list[str]:
|
|
104
|
+
"""Build the hardened ``docker run`` argv (pure; unit-testable without a daemon).
|
|
105
|
+
|
|
106
|
+
Defense-in-depth for the untrusted lane: no network, read-only rootfs, a
|
|
107
|
+
non-root user, ALL capabilities dropped (the bounding set too — uid 65534 only
|
|
108
|
+
clears effective/permitted, but the base image ships setuid-root binaries that
|
|
109
|
+
could otherwise re-escalate), no new privileges, a nosuid size-capped tmpfs as
|
|
110
|
+
the only writable surface, and PID/memory/CPU caps. Per-task, ephemeral
|
|
111
|
+
(``--rm``); args passed as a list (never a shell).
|
|
112
|
+
"""
|
|
113
|
+
|
|
114
|
+
return [
|
|
115
|
+
"docker", "run", "--rm",
|
|
116
|
+
"--name", name,
|
|
117
|
+
"--network", "none", # the real win the subprocess lane can't give
|
|
118
|
+
"--memory", memory,
|
|
119
|
+
"--cpus", cpus,
|
|
120
|
+
"--pids-limit", str(_DEFAULT_PIDS),
|
|
121
|
+
"--user", "65534:65534", # nobody: no in-container root
|
|
122
|
+
"--cap-drop", "ALL", # drop the bounding set (block setuid re-escalation)
|
|
123
|
+
"--security-opt", "no-new-privileges",
|
|
124
|
+
"--read-only", # rootfs read-only; only the tmpfs is writable
|
|
125
|
+
"--tmpfs", "/tmp:size=16m,nosuid", # nosec B108 — container-internal path, not a host temp
|
|
126
|
+
"--entrypoint", "python",
|
|
127
|
+
image,
|
|
128
|
+
"-B", "-c", bootstrap, # -B: no .pyc writes under the read-only fs
|
|
129
|
+
]
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def run_code_tests_docker(
|
|
133
|
+
candidate_code: str,
|
|
134
|
+
checks_code: str,
|
|
135
|
+
*,
|
|
136
|
+
language: str = "python",
|
|
137
|
+
timeout_s: float = 10.0,
|
|
138
|
+
image: str = DEFAULT_IMAGE,
|
|
139
|
+
memory: str = _DEFAULT_MEMORY,
|
|
140
|
+
cpus: str = _DEFAULT_CPUS,
|
|
141
|
+
) -> dict[str, Any]:
|
|
142
|
+
"""Run ``checks_code`` against ``candidate_code`` inside an isolated container.
|
|
143
|
+
|
|
144
|
+
Returns the same ``{"result", "raw"}`` shape as the subprocess verifier; the
|
|
145
|
+
``raw`` block records ``sandbox="docker"``, the image, and isolation flags.
|
|
146
|
+
Never raises for an unavailable daemon or a hostile candidate — both surface
|
|
147
|
+
as an honest failing Result.
|
|
148
|
+
"""
|
|
149
|
+
|
|
150
|
+
if language not in SUPPORTED_LANGUAGES:
|
|
151
|
+
return _empty_result(
|
|
152
|
+
f"unsupported language {language!r}; supported: {SUPPORTED_LANGUAGES}",
|
|
153
|
+
{"sandbox": "docker", "language": language, "infra_error": True},
|
|
154
|
+
)
|
|
155
|
+
if not docker_available():
|
|
156
|
+
return _empty_result(
|
|
157
|
+
"docker unavailable (no daemon / not installed)",
|
|
158
|
+
{"sandbox": "docker", "language": language,
|
|
159
|
+
"docker_available": False, "infra_error": True},
|
|
160
|
+
)
|
|
161
|
+
|
|
162
|
+
name = f"agent-learn-bench-{uuid.uuid4().hex[:12]}"
|
|
163
|
+
raw: dict[str, Any] = {
|
|
164
|
+
"sandbox": "docker",
|
|
165
|
+
"language": language,
|
|
166
|
+
"image": image,
|
|
167
|
+
"network": "none",
|
|
168
|
+
"memory": memory,
|
|
169
|
+
"cpus": cpus,
|
|
170
|
+
"cap_drop": "all",
|
|
171
|
+
"no_new_privileges": True,
|
|
172
|
+
"container": name,
|
|
173
|
+
"timed_out": False,
|
|
174
|
+
"exit_code": None,
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
bootstrap = _DOCKER_BOOTSTRAP.format(
|
|
178
|
+
cand_b64=base64.b64encode(candidate_code.encode("utf-8")).decode("ascii"),
|
|
179
|
+
checks_b64=base64.b64encode(checks_code.encode("utf-8")).decode("ascii"),
|
|
180
|
+
)
|
|
181
|
+
argv = _build_docker_argv(name, image, memory, cpus, bootstrap)
|
|
182
|
+
try:
|
|
183
|
+
proc = subprocess.run(
|
|
184
|
+
argv, capture_output=True, text=True, timeout=timeout_s + 20.0
|
|
185
|
+
)
|
|
186
|
+
except subprocess.TimeoutExpired as exc:
|
|
187
|
+
raw["timed_out"] = True
|
|
188
|
+
raw["stdout_tail"] = _tail(exc.stdout if isinstance(exc.stdout, str) else "")
|
|
189
|
+
raw["stderr_tail"] = _tail(exc.stderr if isinstance(exc.stderr, str) else "")
|
|
190
|
+
_force_kill(name)
|
|
191
|
+
return _empty_result(f"timed out after {timeout_s}s (container killed)", raw)
|
|
192
|
+
|
|
193
|
+
raw["exit_code"] = proc.returncode
|
|
194
|
+
raw["stdout_tail"] = _tail(proc.stdout)
|
|
195
|
+
raw["stderr_tail"] = _tail(proc.stderr)
|
|
196
|
+
|
|
197
|
+
# A daemon/image error (e.g. image not pulled) is infra, not an agent fail.
|
|
198
|
+
if proc.returncode not in (0, 1) and "{" not in (proc.stdout or ""):
|
|
199
|
+
raw["infra_error"] = True
|
|
200
|
+
return _empty_result(
|
|
201
|
+
f"docker run failed (exit {proc.returncode}): {_tail(proc.stderr, 300)}",
|
|
202
|
+
raw,
|
|
203
|
+
)
|
|
204
|
+
|
|
205
|
+
parsed = _parse_runner_stdout(proc.stdout)
|
|
206
|
+
if parsed is None:
|
|
207
|
+
return _empty_result(
|
|
208
|
+
f"runner produced no parseable result (exit {proc.returncode})", raw
|
|
209
|
+
)
|
|
210
|
+
fatal = parsed.get("fatal")
|
|
211
|
+
results = {str(k): bool(v) for k, v in (parsed.get("results") or {}).items()}
|
|
212
|
+
if fatal:
|
|
213
|
+
return _empty_result(str(fatal), raw)
|
|
214
|
+
if not results:
|
|
215
|
+
return _empty_result("no checks executed", raw)
|
|
216
|
+
|
|
217
|
+
total = len(results)
|
|
218
|
+
passed = sum(1 for v in results.values() if v)
|
|
219
|
+
return {
|
|
220
|
+
"result": {
|
|
221
|
+
"scalar": round(passed / total, 6),
|
|
222
|
+
"components": {
|
|
223
|
+
"checks_passed": float(passed),
|
|
224
|
+
"checks_total": float(total),
|
|
225
|
+
},
|
|
226
|
+
"pass_fail": results,
|
|
227
|
+
"explanation": f"{passed}/{total} checks passed",
|
|
228
|
+
},
|
|
229
|
+
"raw": raw,
|
|
230
|
+
}
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
def _force_kill(name: str) -> None:
|
|
234
|
+
try:
|
|
235
|
+
subprocess.run(["docker", "kill", name], capture_output=True, timeout=15)
|
|
236
|
+
except Exception: # nosec B110 — best-effort cleanup; a kill failure must never propagate
|
|
237
|
+
pass
|