agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,144 @@
|
|
|
1
|
+
"""Standing the environment up in containers, with the harness deciding what that means.
|
|
2
|
+
|
|
3
|
+
The harness has read the agent's repository, so it knows what running that agent's code takes:
|
|
4
|
+
which base image, which install command, which store, which services. Encoding any of that here
|
|
5
|
+
would be guessing on behalf of an agent nobody has seen yet, and would be wrong for the next one.
|
|
6
|
+
|
|
7
|
+
So this provides two things and no opinions:
|
|
8
|
+
|
|
9
|
+
- a place to write files, under the session's own ``env`` directory
|
|
10
|
+
- a way to run container commands from there, and read back what happened
|
|
11
|
+
|
|
12
|
+
Everything else, the Dockerfile, the compose file, the schema, the entrypoint, is written by
|
|
13
|
+
whoever read the repository. What is enforced is only what keeps this safe to run on somebody's
|
|
14
|
+
machine: files stay inside the environment directory, and the only commands that run are container
|
|
15
|
+
commands.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import os
|
|
21
|
+
import shlex
|
|
22
|
+
import shutil
|
|
23
|
+
import subprocess
|
|
24
|
+
from pathlib import Path
|
|
25
|
+
|
|
26
|
+
ENV = "env"
|
|
27
|
+
|
|
28
|
+
# Only these. Not a general shell: a tool that can run anything is a tool with no guardrail, and
|
|
29
|
+
# the whole point of routing through here is that what happens is inspectable and bounded.
|
|
30
|
+
ALLOWED = ("docker", "docker-compose")
|
|
31
|
+
|
|
32
|
+
# Long enough for an image build that downloads a base layer, short enough that a hung build is
|
|
33
|
+
# reported rather than waited on forever.
|
|
34
|
+
PATIENCE = 900
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def env_root(destination: Path) -> Path:
|
|
38
|
+
"""Where this agent's environment definition lives, beside its world."""
|
|
39
|
+
root = Path(destination) / ENV
|
|
40
|
+
root.mkdir(parents=True, exist_ok=True)
|
|
41
|
+
return root
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def inside(destination: Path, path: str) -> Path:
|
|
45
|
+
"""The full path for a file the harness wants to write, refused if it escapes.
|
|
46
|
+
|
|
47
|
+
A path arrives as text from a model, so it is resolved and then checked rather than trusted.
|
|
48
|
+
Writing outside the environment directory would mean the harness could touch anything on the
|
|
49
|
+
machine it happens to be running on, which is not a thing to leave to a prompt.
|
|
50
|
+
"""
|
|
51
|
+
root = env_root(destination).resolve()
|
|
52
|
+
asked = (root / str(path).lstrip("/")).resolve()
|
|
53
|
+
if not asked.is_relative_to(root):
|
|
54
|
+
raise ValueError(
|
|
55
|
+
f"{path!r} is outside the environment directory. Everything the environment needs "
|
|
56
|
+
"lives under env/, so that building it cannot reach the rest of the machine."
|
|
57
|
+
)
|
|
58
|
+
return asked
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def write(destination: Path, path: str, contents: str) -> Path:
|
|
62
|
+
"""Put one file into the environment definition."""
|
|
63
|
+
target = inside(destination, path)
|
|
64
|
+
target.parent.mkdir(parents=True, exist_ok=True)
|
|
65
|
+
target.write_text(contents, encoding="utf-8")
|
|
66
|
+
return target
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def listing(destination: Path) -> list[str]:
|
|
70
|
+
root = env_root(destination)
|
|
71
|
+
return sorted(
|
|
72
|
+
str(found.relative_to(root)) for found in root.rglob("*") if found.is_file()
|
|
73
|
+
)
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def available() -> str:
|
|
77
|
+
"""Why containers cannot be used here, or an empty string when they can."""
|
|
78
|
+
if not shutil.which("docker"):
|
|
79
|
+
return "docker is not installed, or not on the path"
|
|
80
|
+
done = subprocess.run(
|
|
81
|
+
["docker", "info", "--format", "{{.ServerVersion}}"],
|
|
82
|
+
capture_output=True,
|
|
83
|
+
text=True,
|
|
84
|
+
timeout=30,
|
|
85
|
+
)
|
|
86
|
+
if done.returncode != 0:
|
|
87
|
+
return (
|
|
88
|
+
f"docker is installed but not running: {(done.stderr or '').strip()[:200]}"
|
|
89
|
+
)
|
|
90
|
+
return ""
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def run(
|
|
94
|
+
destination: Path, command: str, *, patience: int = PATIENCE
|
|
95
|
+
) -> tuple[int, str]:
|
|
96
|
+
"""Run one container command from the environment directory.
|
|
97
|
+
|
|
98
|
+
Returns the exit code and the output, both streams together, because a build failure explains
|
|
99
|
+
itself across the two and reading only one is how the actual cause gets lost.
|
|
100
|
+
"""
|
|
101
|
+
try:
|
|
102
|
+
words = shlex.split(command)
|
|
103
|
+
except ValueError as exc:
|
|
104
|
+
return 1, f"could not parse container command: {exc}"
|
|
105
|
+
if not words:
|
|
106
|
+
return 1, "no command given"
|
|
107
|
+
if words[0] not in ALLOWED:
|
|
108
|
+
return 1, (
|
|
109
|
+
f"{words[0]!r} is not something this can run. Only {' and '.join(ALLOWED)} commands, "
|
|
110
|
+
"because a general shell here would be a guardrail with nothing behind it. Everything "
|
|
111
|
+
"the environment needs should be in a file it builds from, not in a command."
|
|
112
|
+
)
|
|
113
|
+
blocked = available()
|
|
114
|
+
if blocked:
|
|
115
|
+
return 1, blocked
|
|
116
|
+
# When the daemon is remote (DOCKER_HOST at a socket proxy), a bind mount
|
|
117
|
+
# names a path on the daemon's host — this container's own filesystem is
|
|
118
|
+
# invisible to it. The mount comes up empty and the failure reads as a
|
|
119
|
+
# missing file three steps later, so it is refused here with the reason.
|
|
120
|
+
if os.environ.get("DOCKER_HOST") and (
|
|
121
|
+
" -v " in f" {command} " or "--volume" in command
|
|
122
|
+
):
|
|
123
|
+
return 1, (
|
|
124
|
+
"bind mounts cannot work in this deployment: the docker daemon runs "
|
|
125
|
+
"outside this container and does not see these paths. Run the script "
|
|
126
|
+
"inline instead (sh -c '<script>'), or COPY files into an image with "
|
|
127
|
+
"a Dockerfile — build contexts do transfer."
|
|
128
|
+
)
|
|
129
|
+
try:
|
|
130
|
+
done = subprocess.run(
|
|
131
|
+
words,
|
|
132
|
+
cwd=str(env_root(destination)),
|
|
133
|
+
capture_output=True,
|
|
134
|
+
text=True,
|
|
135
|
+
timeout=patience,
|
|
136
|
+
)
|
|
137
|
+
except subprocess.TimeoutExpired:
|
|
138
|
+
return 1, (
|
|
139
|
+
f"gave up after {patience}s. An install that takes this long usually means a "
|
|
140
|
+
"dependency is being fetched that is not going to arrive; check what the last step "
|
|
141
|
+
"was trying to reach."
|
|
142
|
+
)
|
|
143
|
+
output = ((done.stdout or "") + (done.stderr or "")).strip()
|
|
144
|
+
return done.returncode, output
|
fi/alk/image_loop.py
ADDED
|
@@ -0,0 +1,453 @@
|
|
|
1
|
+
"""Phase 9B units 1-4 — the image / multimodal improvement loop (the 13D
|
|
2
|
+
Practice Loop on ``world.kind = image``).
|
|
3
|
+
|
|
4
|
+
ARCH-9B §2.1/§2.2/§2.3/§2.4 / decisions 9B-D1..9B-D6, 9B-A1/A2/A3/A7/A8.
|
|
5
|
+
|
|
6
|
+
This module invents NO optimizer, NO artifact kind, NO loss machinery, NO world.
|
|
7
|
+
It is the IMAGE analogue of ``voice_loop.py`` — a thin composition layer over
|
|
8
|
+
verbatim engines:
|
|
9
|
+
|
|
10
|
+
* the multi-objective image loss compiles via ``loss.compile_objective`` (the
|
|
11
|
+
Goodhart guard at ``loss.py:106-116`` is reused VERBATIM — "There is no
|
|
12
|
+
override."); the 9B-A2 composition rule (>= 2 terms, >= 1 deterministic
|
|
13
|
+
ground-truth anchor — a judge-only loss is INVALID) is a thin validator on
|
|
14
|
+
top, raising ``image_loss_guard_missing`` (``ImageLossCompositionError``);
|
|
15
|
+
* the whole multimodal-agent config is the search space, assembled by
|
|
16
|
+
``optimize.build_practice_loop_manifest`` (the same ``base_agent`` +
|
|
17
|
+
``search_space`` whole-agent contract) with ``world.kind=image`` +
|
|
18
|
+
``task_mode`` (understanding | generation) on ``WorldSpec.spec``;
|
|
19
|
+
* the image sub-attribution is an additive tag stamped alongside the base
|
|
20
|
+
``FAILURE_LAYERS`` tag (the existing ``practice/_diagnose.py`` machinery is
|
|
21
|
+
consumed, not rewritten);
|
|
22
|
+
* ``world.kind=image`` enters the world-kind space through the R4 registry hook
|
|
23
|
+
(``extensions.register_extension``) — never by widening the frozen
|
|
24
|
+
``SIMULATION_WORLD_KINDS`` tuple. ``image`` is "typed -> executable".
|
|
25
|
+
|
|
26
|
+
The canon constants below are this module's home; ``trinity.py`` carries literal
|
|
27
|
+
mirrors that the milestone test cross-pins (the GUNA_AXES cross-pin pattern —
|
|
28
|
+
trinity never imports this module so the gate runs even if this is broken). The
|
|
29
|
+
pure-numpy perturbation operators live in the companion ``image_perturb.py``
|
|
30
|
+
(9B-A1b — substrate, not loop).
|
|
31
|
+
"""
|
|
32
|
+
|
|
33
|
+
from __future__ import annotations
|
|
34
|
+
|
|
35
|
+
from typing import Any, Mapping, Optional, Sequence
|
|
36
|
+
|
|
37
|
+
# --- canon (ARCH-9B §2.1 image-loss term refs + §2.3 sub-attribution) -------
|
|
38
|
+
# The deterministic-anchored UNDERSTANDING-mode menu (the 6-tuple analogue of
|
|
39
|
+
# V1_VOICE_LOSS_TERM_REFS, the 9-tuple).
|
|
40
|
+
V1_IMAGE_LOSS_TERM_REFS = (
|
|
41
|
+
"task_success",
|
|
42
|
+
"ocr_accuracy",
|
|
43
|
+
"chart_accuracy",
|
|
44
|
+
"artifact_grounding",
|
|
45
|
+
"instruction_adherence",
|
|
46
|
+
"tool_argument_correctness",
|
|
47
|
+
)
|
|
48
|
+
# The mandatory ground-truth quality anchors — an image loss MUST carry >= 1 of
|
|
49
|
+
# these (9B-A2 / 9B-D3). The analogue of V1_VOICE_LOSS_NON_TIMING_QUALITY_TERMS.
|
|
50
|
+
# ``element_presence`` joins this admissible set under task_mode=generation.
|
|
51
|
+
V1_IMAGE_LOSS_DETERMINISTIC_ANCHOR_TERMS = (
|
|
52
|
+
"task_success",
|
|
53
|
+
"ocr_accuracy",
|
|
54
|
+
"chart_accuracy",
|
|
55
|
+
"artifact_grounding",
|
|
56
|
+
)
|
|
57
|
+
# The bounded/guarded judge contributors (the analogue of V1_VOICE_LOSS_TIMING_
|
|
58
|
+
# TERMS, the hackable-alone set). A judge-only loss (terms subset of this set) is
|
|
59
|
+
# structurally rejected (9B-D3). generation_alignment / generation_quality join
|
|
60
|
+
# this set under task_mode=generation.
|
|
61
|
+
V1_IMAGE_LOSS_JUDGE_TERMS = ("instruction_adherence",)
|
|
62
|
+
|
|
63
|
+
# Generation-mode terms (ARCH-9B §2.4 / 9B-A7/A8), admitted ONLY under
|
|
64
|
+
# task_mode=generation.
|
|
65
|
+
V1_IMAGE_GENERATION_ANCHOR_TERMS = ("element_presence",) # deterministic floor (9B-A8)
|
|
66
|
+
V1_IMAGE_GENERATION_JUDGE_TERMS = ("generation_alignment", "generation_quality")
|
|
67
|
+
|
|
68
|
+
# The four-token image sub-attribution closed set (9B §2.3), stamped alongside
|
|
69
|
+
# the base FAILURE_LAYERS tag.
|
|
70
|
+
V1_IMAGE_FAILURE_SUBLAYERS = ("preprocessing", "perception", "reasoning", "tool_grounding")
|
|
71
|
+
|
|
72
|
+
# A MARKER field on artifact metadata — NOT a new evidence class (R5/A18; the
|
|
73
|
+
# frozen EVIDENCE_CLASSES 4-tuple is unchanged). The analogue of
|
|
74
|
+
# V1_VOICE_FIDELITY_TIERS. (ARCH-9B §2.6)
|
|
75
|
+
V1_IMAGE_FIDELITY_TIERS = ("deterministic_fixture", "keyed_live_model")
|
|
76
|
+
|
|
77
|
+
# The typed ``kind`` discriminators a perception-bypass guard row may carry,
|
|
78
|
+
# beyond the base sentinel/canary rows the loss guard already allows (ARCH-9B
|
|
79
|
+
# §2.2).
|
|
80
|
+
V1_IMAGE_PERCEPTION_GUARD_KINDS = ("perception_bypass", "perceptual_counterfactual")
|
|
81
|
+
|
|
82
|
+
# The task_mode switch on WorldSpec.spec (ARCH-9B §2.3 / 9B-D2). ONE world kind,
|
|
83
|
+
# two loss profiles.
|
|
84
|
+
V1_IMAGE_TASK_MODES = ("understanding", "generation")
|
|
85
|
+
|
|
86
|
+
# The registered world-kind token + the namespaced extension name (R4 hook).
|
|
87
|
+
IMAGE_WORLD_KIND = "image"
|
|
88
|
+
IMAGE_EXTENSION_NAME = "agentlearning.image"
|
|
89
|
+
|
|
90
|
+
# The R4 rung -> evidence-class ladder (ARCH-9B §2.6). The deterministic core is
|
|
91
|
+
# local_gate/captured_fixture; live_lane is added ONLY on the keyed lane record
|
|
92
|
+
# (unit 7), never the day-one deterministic record.
|
|
93
|
+
_IMAGE_RUNG_LADDER = {
|
|
94
|
+
"rung1": ["local_gate"],
|
|
95
|
+
"perturbed": ["live_stressed", "captured_fixture"],
|
|
96
|
+
"keyed_vlm": ["live_lane"],
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
class ImageLossCompositionError(ValueError):
|
|
101
|
+
"""Raised when an image objective violates the 9B-A2 composition rule (the
|
|
102
|
+
``image_loss_guard_missing`` finding — an image specialization of
|
|
103
|
+
``objective_guards_missing``). A ``ValueError`` subclass so callers can
|
|
104
|
+
``except ValueError`` exactly as for ``VoiceLossCompositionError``."""
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def _term_refs(objective: Mapping[str, Any]) -> list[str]:
|
|
108
|
+
"""The objective's eval refs (read from ``evals`` — the loss.py schema)."""
|
|
109
|
+
return [
|
|
110
|
+
str(term.get("eval"))
|
|
111
|
+
for term in (objective.get("evals") or [])
|
|
112
|
+
if isinstance(term, Mapping) and term.get("eval")
|
|
113
|
+
]
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def _admissible_anchor_terms(task_mode: str) -> tuple[str, ...]:
|
|
117
|
+
if task_mode == "generation":
|
|
118
|
+
return V1_IMAGE_LOSS_DETERMINISTIC_ANCHOR_TERMS + V1_IMAGE_GENERATION_ANCHOR_TERMS
|
|
119
|
+
return V1_IMAGE_LOSS_DETERMINISTIC_ANCHOR_TERMS
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def _admissible_term_refs(task_mode: str) -> tuple[str, ...]:
|
|
123
|
+
if task_mode == "generation":
|
|
124
|
+
return (
|
|
125
|
+
V1_IMAGE_LOSS_TERM_REFS
|
|
126
|
+
+ V1_IMAGE_GENERATION_ANCHOR_TERMS
|
|
127
|
+
+ V1_IMAGE_GENERATION_JUDGE_TERMS
|
|
128
|
+
)
|
|
129
|
+
return V1_IMAGE_LOSS_TERM_REFS
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def compile_image_objective(
|
|
133
|
+
payload: Mapping[str, Any], *, task_mode: str = "understanding"
|
|
134
|
+
) -> dict:
|
|
135
|
+
"""Compile a multi-objective image loss with a perception-bypass Goodhart
|
|
136
|
+
guard (ARCH-9B §2.2 / 9B-A2 / 9B-D3). The image analogue of
|
|
137
|
+
``compile_voice_objective`` (voice_loop.py:70). Enforces, ON TOP of the
|
|
138
|
+
verbatim ``loss.compile_objective`` Goodhart guard:
|
|
139
|
+
|
|
140
|
+
(a) >= 2 terms (a single-term image objective is reward-hackable);
|
|
141
|
+
(b) >= 1 deterministic ground-truth anchor — a judge-only loss is INVALID
|
|
142
|
+
(9B-D3). ``task_mode`` selects the admissible anchor set:
|
|
143
|
+
understanding -> V1_IMAGE_LOSS_DETERMINISTIC_ANCHOR_TERMS; generation
|
|
144
|
+
adds ``element_presence`` (the deterministic floor, 9B-A8);
|
|
145
|
+
(c) unknown-ref rejection (every term must be a member of the mode's menu);
|
|
146
|
+
(d) when sentinel/canary rows carry a perception ``kind`` discriminator it
|
|
147
|
+
must be in V1_IMAGE_PERCEPTION_GUARD_KINDS (the closed set).
|
|
148
|
+
|
|
149
|
+
Then delegates to ``loss.compile_objective`` VERBATIM — which unconditionally
|
|
150
|
+
enforces the populated guard block (sentinel_rows / canary_evals,
|
|
151
|
+
min_guard_count >= 1, "There is no override.")."""
|
|
152
|
+
|
|
153
|
+
from . import loss as _loss # downward facade import (legal; voice_loop.py idiom)
|
|
154
|
+
|
|
155
|
+
if task_mode not in V1_IMAGE_TASK_MODES:
|
|
156
|
+
raise ImageLossCompositionError(
|
|
157
|
+
f"image_loss_guard_missing: task_mode {task_mode!r} not in "
|
|
158
|
+
f"{V1_IMAGE_TASK_MODES}"
|
|
159
|
+
)
|
|
160
|
+
|
|
161
|
+
refs = _term_refs(payload)
|
|
162
|
+
|
|
163
|
+
# rule (a): >= 2 terms.
|
|
164
|
+
if len(refs) < 2:
|
|
165
|
+
raise ImageLossCompositionError(
|
|
166
|
+
"image_loss_guard_missing: an image objective is reward-hackable as a "
|
|
167
|
+
"single term; it MUST be multi-objective (>= 2 terms). "
|
|
168
|
+
f"got {refs}"
|
|
169
|
+
)
|
|
170
|
+
|
|
171
|
+
# rule (b): >= 1 deterministic ground-truth anchor (judge-only REJECTED).
|
|
172
|
+
anchors = _admissible_anchor_terms(task_mode)
|
|
173
|
+
if not any(ref in anchors for ref in refs):
|
|
174
|
+
raise ImageLossCompositionError(
|
|
175
|
+
"image_loss_guard_missing: an image loss MUST carry >= 1 deterministic "
|
|
176
|
+
f"ground-truth anchor {anchors}; a judge-only loss is INVALID by "
|
|
177
|
+
f"contract (9B-D3). got {refs}"
|
|
178
|
+
)
|
|
179
|
+
|
|
180
|
+
# rule (c): unknown-ref rejection.
|
|
181
|
+
allowed = _admissible_term_refs(task_mode)
|
|
182
|
+
for ref in refs:
|
|
183
|
+
if ref not in allowed:
|
|
184
|
+
raise ImageLossCompositionError(
|
|
185
|
+
f"image_loss_guard_missing: unknown image loss term {ref!r}; "
|
|
186
|
+
f"expected members of {allowed} (task_mode={task_mode})"
|
|
187
|
+
)
|
|
188
|
+
|
|
189
|
+
# rule (d): the perception-bypass guard rows ride the existing
|
|
190
|
+
# sentinel_rows/canary_evals with a typed ``kind`` discriminator (no new
|
|
191
|
+
# ObjectiveSpec field, ARCH-9B §2.2). When present it must be in the closed
|
|
192
|
+
# set (plus any untyped/base rows the loss guard already allows).
|
|
193
|
+
guards = payload.get("guards") or {}
|
|
194
|
+
for bucket in ("sentinel_rows", "canary_evals"):
|
|
195
|
+
for row in guards.get(bucket) or []:
|
|
196
|
+
if isinstance(row, Mapping):
|
|
197
|
+
kind = row.get("kind")
|
|
198
|
+
if kind is not None and kind not in V1_IMAGE_PERCEPTION_GUARD_KINDS:
|
|
199
|
+
raise ImageLossCompositionError(
|
|
200
|
+
f"image_loss_guard_missing: guard row kind {kind!r} not in "
|
|
201
|
+
f"{V1_IMAGE_PERCEPTION_GUARD_KINDS}"
|
|
202
|
+
)
|
|
203
|
+
|
|
204
|
+
# the verbatim Goodhart guard (loss.py:106-116) — "There is no override."
|
|
205
|
+
return _loss.compile_objective(payload)
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
def attribute_image_sublayer(
|
|
209
|
+
*,
|
|
210
|
+
failure_layer: str,
|
|
211
|
+
deficit: Mapping[str, Any] | None = None,
|
|
212
|
+
signal: str | None = None,
|
|
213
|
+
) -> str:
|
|
214
|
+
"""Map a weak image cell to a ``V1_IMAGE_FAILURE_SUBLAYERS`` token, stamped
|
|
215
|
+
ALONGSIDE the base ``FAILURE_LAYERS`` tag (a weak cell carries both, e.g.
|
|
216
|
+
``{failure_layer:"agent_behavior", image_sublayer:"perception"}``). The base
|
|
217
|
+
attribution rides the existing ``practice/_diagnose.py`` machinery; this is
|
|
218
|
+
the thin sublayer helper (the image analogue of ``attribute_voice_sublayer``,
|
|
219
|
+
voice_loop.py:109).
|
|
220
|
+
|
|
221
|
+
Routing (ARCH-9B §2.3 table, grounded in DISCO's parse->reason split
|
|
222
|
+
2603.23511 + AgentVista visual-misidentification=perception 2602.23166):
|
|
223
|
+
* OCR/parse weak; low-res/compression hurts -> ``preprocessing``
|
|
224
|
+
(Fix-Before-Search: try a resolution/crop change before blaming the LLM);
|
|
225
|
+
* visual misidentification; perception-required cell weak -> ``perception``
|
|
226
|
+
(the dominant AgentVista failure);
|
|
227
|
+
* grounded-but-wrong-conclusion -> ``reasoning`` (parse ok, reasoning fails);
|
|
228
|
+
* tool-argument extracted wrong from the image -> ``tool_grounding``."""
|
|
229
|
+
|
|
230
|
+
sig = str(signal or (deficit or {}).get("signal") or "").lower()
|
|
231
|
+
if any(
|
|
232
|
+
k in sig
|
|
233
|
+
for k in ("ocr", "parse", "low_res", "low-res", "resolution", "compression",
|
|
234
|
+
"compress", "blur", "preprocess")
|
|
235
|
+
):
|
|
236
|
+
return "preprocessing"
|
|
237
|
+
if any(
|
|
238
|
+
k in sig
|
|
239
|
+
for k in ("tool_argument", "tool-argument", "tool_grounding", "tool argument",
|
|
240
|
+
"argument", "extracted")
|
|
241
|
+
):
|
|
242
|
+
return "tool_grounding"
|
|
243
|
+
if any(
|
|
244
|
+
k in sig
|
|
245
|
+
for k in ("misidentif", "perception", "visual", "occlusion", "occluded",
|
|
246
|
+
"perceive", "see ")
|
|
247
|
+
):
|
|
248
|
+
return "perception"
|
|
249
|
+
if any(
|
|
250
|
+
k in sig
|
|
251
|
+
for k in ("reason", "conclusion", "grounded-but-wrong", "wrong_conclusion",
|
|
252
|
+
"inference")
|
|
253
|
+
):
|
|
254
|
+
return "reasoning"
|
|
255
|
+
# default: infra-implicated cells land on preprocessing (the cheapest fix
|
|
256
|
+
# before blaming the model); otherwise the reasoning/policy layer.
|
|
257
|
+
if failure_layer in ("lane_infra", "framework_runtime", "provider"):
|
|
258
|
+
return "preprocessing"
|
|
259
|
+
return "reasoning"
|
|
260
|
+
|
|
261
|
+
|
|
262
|
+
def _ensure_image_world_registered() -> None:
|
|
263
|
+
"""Register ``world.kind=image`` via the R4 hook (ARCH-9B §2.1 / 9B-D2).
|
|
264
|
+
Idempotent. Pushes DOWN into ``contract.register_world_kind`` so
|
|
265
|
+
``resolved_world_kinds()`` contains ``image`` WITHOUT touching the frozen
|
|
266
|
+
``SIMULATION_WORLD_KINDS`` tuple (contract.py:55)."""
|
|
267
|
+
|
|
268
|
+
from fi.simulate.simulation import contract as _contract
|
|
269
|
+
|
|
270
|
+
# idempotent: register_extension raises on a name collision, and the world
|
|
271
|
+
# kind only needs to land once per process.
|
|
272
|
+
if IMAGE_WORLD_KIND in _contract.resolved_world_kinds():
|
|
273
|
+
return
|
|
274
|
+
|
|
275
|
+
from . import extensions as _ext
|
|
276
|
+
|
|
277
|
+
_ext.register_extension(
|
|
278
|
+
"environment",
|
|
279
|
+
{
|
|
280
|
+
"name": IMAGE_EXTENSION_NAME, # vendor.name shape (_validate_record)
|
|
281
|
+
"kind_token": IMAGE_WORLD_KIND, # the registered world.kind token
|
|
282
|
+
"spec_validator": _validate_image_world_spec, # R4 mandate
|
|
283
|
+
"rung_ladder": _IMAGE_RUNG_LADDER, # R4 mandate
|
|
284
|
+
# the deterministic core; live_lane is added ONLY on the keyed lane
|
|
285
|
+
# record (unit 7), never here.
|
|
286
|
+
"evidence_class_capability": ["local_gate", "captured_fixture"],
|
|
287
|
+
# gated_contexts_runnable stays False until rung1_fixture_green
|
|
288
|
+
# (extensions.py admission); 9B never silently claims executable.
|
|
289
|
+
},
|
|
290
|
+
)
|
|
291
|
+
|
|
292
|
+
|
|
293
|
+
def _validate_image_world_spec(spec: Mapping[str, Any]) -> None:
|
|
294
|
+
"""The R4 ``spec_validator`` for the image world: validate the ``task_mode``
|
|
295
|
+
switch (understanding | generation) on ``WorldSpec.spec``. Raises ValueError
|
|
296
|
+
on an unknown mode (the closed-set guard)."""
|
|
297
|
+
|
|
298
|
+
task_mode = str((spec or {}).get("task_mode", "understanding"))
|
|
299
|
+
if task_mode not in V1_IMAGE_TASK_MODES:
|
|
300
|
+
raise ValueError(
|
|
301
|
+
f"image world.spec.task_mode {task_mode!r} not in {V1_IMAGE_TASK_MODES}"
|
|
302
|
+
)
|
|
303
|
+
|
|
304
|
+
|
|
305
|
+
def build_image_practice_loop_manifest(
|
|
306
|
+
*,
|
|
307
|
+
name: str,
|
|
308
|
+
base_agent: Mapping[str, Any],
|
|
309
|
+
search_space: Mapping[str, Sequence[Any]],
|
|
310
|
+
objective: Mapping[str, Any],
|
|
311
|
+
eval_budget: int,
|
|
312
|
+
seed: int,
|
|
313
|
+
task_mode: str = "understanding",
|
|
314
|
+
scenario_inline: Optional[Mapping[str, Any]] = None,
|
|
315
|
+
max_rounds: int = 8,
|
|
316
|
+
) -> dict[str, Any]:
|
|
317
|
+
"""Assemble the image improvement-loop manifest: the 13D Practice Loop on
|
|
318
|
+
``world.kind=image`` + ``task_mode`` with the multi-objective guarded image
|
|
319
|
+
loss + the whole multimodal-agent search space (9B-D5). Delegates to
|
|
320
|
+
``optimize.build_practice_loop_manifest`` so its validators hold VERBATIM
|
|
321
|
+
(9B-A3). The objective is compiled by ``compile_image_objective`` (the 9B-A2
|
|
322
|
+
rule) before it rides the simulation.
|
|
323
|
+
|
|
324
|
+
Byte-parallel to ``build_voice_practice_loop_manifest`` except: (a) the
|
|
325
|
+
``_ensure_image_world_registered()`` call (voice's kind is built-in, image's
|
|
326
|
+
is registered through the R4 hook); (b) ``world["kind"]="image"`` instead of
|
|
327
|
+
``"voice_telephony"``; (c) the ``spec["task_mode"]`` write (voice has no mode
|
|
328
|
+
switch)."""
|
|
329
|
+
|
|
330
|
+
from . import optimize as _optimize # downward facade import (legal)
|
|
331
|
+
|
|
332
|
+
if task_mode not in V1_IMAGE_TASK_MODES:
|
|
333
|
+
raise ImageLossCompositionError(
|
|
334
|
+
f"image_loss_guard_missing: task_mode {task_mode!r} not in "
|
|
335
|
+
f"{V1_IMAGE_TASK_MODES}"
|
|
336
|
+
)
|
|
337
|
+
|
|
338
|
+
_ensure_image_world_registered() # step 1 (§2.1)
|
|
339
|
+
compiled = compile_image_objective(objective, task_mode=task_mode) # step 2 (unit 2)
|
|
340
|
+
inline = dict(scenario_inline or {})
|
|
341
|
+
inline.setdefault("version", "agent-learning.simulation.v1")
|
|
342
|
+
inline["objective"] = compiled
|
|
343
|
+
world = dict(inline.get("world") or {})
|
|
344
|
+
world["kind"] = IMAGE_WORLD_KIND # step 3 — the registered kind
|
|
345
|
+
spec = dict(world.get("spec") or {})
|
|
346
|
+
spec["task_mode"] = task_mode # the task_mode switch on WorldSpec.spec
|
|
347
|
+
world["spec"] = spec
|
|
348
|
+
inline["world"] = world
|
|
349
|
+
|
|
350
|
+
return _optimize.build_practice_loop_manifest( # step 4 — VERBATIM delegate
|
|
351
|
+
name=name,
|
|
352
|
+
simulation={"version": inline["version"], "inline": inline},
|
|
353
|
+
base_agent=base_agent,
|
|
354
|
+
search_space=search_space,
|
|
355
|
+
eval_budget=eval_budget,
|
|
356
|
+
seed=seed,
|
|
357
|
+
max_rounds=max_rounds,
|
|
358
|
+
)
|
|
359
|
+
|
|
360
|
+
|
|
361
|
+
# === Unit 7 — the keyed real-VLM lane (opt-in, NEVER a gate prerequisite) ===
|
|
362
|
+
# ARCH-9B §2.4 / §2.6 / 9B-D1/D6. The judge-anchored terms, the full generation
|
|
363
|
+
# profile, and the one real-multimodal-agent live-proof are owner-keyed, opt-in,
|
|
364
|
+
# never a release gate. The deterministic core stays local_gate/captured_fixture;
|
|
365
|
+
# the keyed lane is the ONLY honest place for live_lane (a real keyed model ran).
|
|
366
|
+
|
|
367
|
+
KEYED_IMAGE_EXTENSION_NAME = "agentlearning.image.keyed"
|
|
368
|
+
_KEYED_IMAGE_RUNG_LADDER = {
|
|
369
|
+
"rung1": ["local_gate"],
|
|
370
|
+
"perturbed": ["live_stressed", "captured_fixture"],
|
|
371
|
+
"keyed_vlm": ["live_lane"],
|
|
372
|
+
}
|
|
373
|
+
|
|
374
|
+
# the env keys that gate the keyed lane (checked, never required by any gate).
|
|
375
|
+
IMAGE_JUDGE_KEY_ENVS = ("AGENT_LEARNING_IMAGE_JUDGE_KEY", "OPENAI_API_KEY")
|
|
376
|
+
|
|
377
|
+
|
|
378
|
+
class ImageKeyedLaneUnavailable(RuntimeError):
|
|
379
|
+
"""Raised by the keyed lane when no judge/VLM key is present — the loud
|
|
380
|
+
refusal (the ``image_judge_key_unavailable`` finding). The deterministic core
|
|
381
|
+
NEVER raises this; only the opt-in keyed path does."""
|
|
382
|
+
|
|
383
|
+
|
|
384
|
+
def image_judge_key_present() -> bool:
|
|
385
|
+
"""True iff a judge/VLM key is configured for the keyed lane."""
|
|
386
|
+
import os
|
|
387
|
+
|
|
388
|
+
return any(os.environ.get(env) for env in IMAGE_JUDGE_KEY_ENVS)
|
|
389
|
+
|
|
390
|
+
|
|
391
|
+
def register_keyed_image_lane() -> None:
|
|
392
|
+
"""Register the SEPARATE keyed-lane extension record that adds ``live_lane``
|
|
393
|
+
to ``evidence_class_capability`` (ARCH-9B §2.6, unit 7). Idempotent. This is
|
|
394
|
+
the ONLY record that may carry ``live_lane`` — the deterministic-core record
|
|
395
|
+
(``_ensure_image_world_registered``) stays ``("local_gate","captured_fixture")``.
|
|
396
|
+
NEVER called by the gate; opt-in only."""
|
|
397
|
+
from . import extensions as _ext
|
|
398
|
+
|
|
399
|
+
if _ext.resolve("environment", KEYED_IMAGE_EXTENSION_NAME) is not None:
|
|
400
|
+
return
|
|
401
|
+
_ext.register_extension(
|
|
402
|
+
"environment",
|
|
403
|
+
{
|
|
404
|
+
"name": KEYED_IMAGE_EXTENSION_NAME,
|
|
405
|
+
# the keyed lane reuses the SAME world.kind token only when the base
|
|
406
|
+
# record is absent; here it declares the keyed capability without a
|
|
407
|
+
# second kind_token (the world kind is already registered).
|
|
408
|
+
"evidence_class_capability": ["local_gate", "captured_fixture", "live_lane"],
|
|
409
|
+
},
|
|
410
|
+
)
|
|
411
|
+
|
|
412
|
+
|
|
413
|
+
def run_keyed_image_live_proof(
|
|
414
|
+
*,
|
|
415
|
+
base_agent: Mapping[str, Any],
|
|
416
|
+
search_space: Mapping[str, Sequence[Any]],
|
|
417
|
+
objective: Mapping[str, Any],
|
|
418
|
+
eval_budget: int,
|
|
419
|
+
seed: int,
|
|
420
|
+
task_mode: str = "generation",
|
|
421
|
+
name: str = "image-keyed-live-proof",
|
|
422
|
+
) -> dict[str, Any]:
|
|
423
|
+
"""The one owner-keyed live-proof entry (WORKFLOW Step 5 real-keys ground
|
|
424
|
+
rule). Refuses LOUDLY without a key (``ImageKeyedLaneUnavailable`` ->
|
|
425
|
+
``image_judge_key_unavailable``) — never a fake number, never a release
|
|
426
|
+
prerequisite. With a key, it builds the generation-profile manifest and marks
|
|
427
|
+
the run ``live_lane`` / ``fidelity_tier=keyed_live_model``.
|
|
428
|
+
|
|
429
|
+
The keyed run itself (calling the judge/VLM) is left to the caller's runtime;
|
|
430
|
+
this returns the keyed manifest + the honest evidence-class stamp so an
|
|
431
|
+
owner can execute it once with real keys."""
|
|
432
|
+
if not image_judge_key_present():
|
|
433
|
+
raise ImageKeyedLaneUnavailable(
|
|
434
|
+
"image_judge_key_unavailable: the keyed real-VLM lane requires a "
|
|
435
|
+
f"judge/VLM key (one of {IMAGE_JUDGE_KEY_ENVS}); withheld -- never a "
|
|
436
|
+
"fake number, never a release prerequisite"
|
|
437
|
+
)
|
|
438
|
+
register_keyed_image_lane()
|
|
439
|
+
manifest = build_image_practice_loop_manifest(
|
|
440
|
+
name=name,
|
|
441
|
+
base_agent=base_agent,
|
|
442
|
+
search_space=search_space,
|
|
443
|
+
objective=objective,
|
|
444
|
+
eval_budget=eval_budget,
|
|
445
|
+
seed=seed,
|
|
446
|
+
task_mode=task_mode,
|
|
447
|
+
)
|
|
448
|
+
return {
|
|
449
|
+
"manifest": manifest,
|
|
450
|
+
"evidence_class": "live_lane", # the ONLY honest live_lane
|
|
451
|
+
"fidelity_tier": "keyed_live_model",
|
|
452
|
+
"task_mode": task_mode,
|
|
453
|
+
}
|