agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
fi/alk/cua_loop.py
ADDED
|
@@ -0,0 +1,562 @@
|
|
|
1
|
+
"""Phase 9C units 1-4 — the CUA / browser / computer-use improvement loop (the
|
|
2
|
+
13D Practice Loop on ``world.kind = browser`` / ``computer_use`` with a
|
|
3
|
+
``cua_surface = browser | desktop`` sub-kind switch).
|
|
4
|
+
|
|
5
|
+
ARCH-9C §2.1/§2.2/§2.3/§2.4 / decisions 9C-D1..9C-D6, 9C-A1/A1b/A1c/A2/A3/A7/A7b/A8.
|
|
6
|
+
|
|
7
|
+
This module invents NO optimizer, NO artifact kind, NO loss machinery, NO world,
|
|
8
|
+
NO perturbation module. It is the CUA analogue of ``image_loop.py`` /
|
|
9
|
+
``voice_loop.py`` — a thin composition layer over verbatim engines:
|
|
10
|
+
|
|
11
|
+
* the multi-objective CUA loss compiles via ``loss.compile_objective`` (the
|
|
12
|
+
Goodhart guard at ``loss.py:106-116`` is reused VERBATIM — "There is no
|
|
13
|
+
override."); the 9C-A2 composition rule (>= 2 terms, >= 1 deterministic
|
|
14
|
+
post-state ground-truth anchor — a judge-only loss is INVALID) is a thin
|
|
15
|
+
validator on top, raising ``cua_loss_guard_missing`` (``CuaLossCompositionError``);
|
|
16
|
+
* the whole CUA-agent config is the search space, assembled by
|
|
17
|
+
``optimize.build_practice_loop_manifest`` (the same ``base_agent`` +
|
|
18
|
+
``search_space`` whole-agent contract) with ``world.kind=browser`` /
|
|
19
|
+
``computer_use`` + ``cua_surface`` (browser | desktop) on ``WorldSpec.spec``;
|
|
20
|
+
* the CUA sub-attribution is an additive tag stamped alongside the base
|
|
21
|
+
``FAILURE_LAYERS`` tag (the existing ``practice/_diagnose.py`` machinery is
|
|
22
|
+
consumed, not rewritten);
|
|
23
|
+
* ``browser`` / ``computer_use`` enter EXECUTABLE-LOOP status through the R4
|
|
24
|
+
registry hook (``extensions.register_extension``) — never by widening the
|
|
25
|
+
frozen ``SIMULATION_WORLD_KINDS`` tuple. They are already FROZEN typed-only
|
|
26
|
+
members (the §0/9C-A1b nuance vs 9B's ``image``): so the registration gates on
|
|
27
|
+
the EXECUTABLE-LOOP RECORD presence in ``_EXTRA_WORLD_KINDS``, NOT on the
|
|
28
|
+
verbatim image idempotence guard (which would short-circuit immediately).
|
|
29
|
+
|
|
30
|
+
The CUA perturbation operators (selector-drift / layout-shift / stale-screenshot
|
|
31
|
+
/ injected-DOM) are ALREADY in the kit's ``BrowserEnvironment`` mutation pack
|
|
32
|
+
(9C-A1c / 9C-D4) — there is NO ``cua_perturb.py`` and NO ``apply_cua_perturbations``
|
|
33
|
+
function (the sharpest contrast with 9B's ``image_perturb.py``);
|
|
34
|
+
``V1_CUA_PERTURBATION_OPERATORS`` is a NAMING MIRROR ONLY.
|
|
35
|
+
|
|
36
|
+
The canon constants below are this module's home; ``trinity.py`` carries literal
|
|
37
|
+
mirrors that the milestone test cross-pins (the GUNA_AXES cross-pin pattern —
|
|
38
|
+
trinity never imports this module so the gate runs even if this is broken).
|
|
39
|
+
"""
|
|
40
|
+
|
|
41
|
+
from __future__ import annotations
|
|
42
|
+
|
|
43
|
+
from typing import Any, Mapping, Optional, Sequence
|
|
44
|
+
|
|
45
|
+
# --- canon (ARCH-9C §2.1 CUA-loss term refs + §2.3 sub-attribution) ----------
|
|
46
|
+
# The browser-surface loss menu (the 9-tuple analogue of V1_IMAGE_LOSS_TERM_REFS,
|
|
47
|
+
# the 6-tuple). ``grounding_step_accuracy`` is admitted only under
|
|
48
|
+
# cua_surface=desktop — see V1_CUA_DESKTOP_ANCHOR_TERMS / unit 4.
|
|
49
|
+
V1_CUA_LOSS_TERM_REFS = (
|
|
50
|
+
"task_success",
|
|
51
|
+
"state_match",
|
|
52
|
+
"grounding_mutation_resilience",
|
|
53
|
+
"action_correctness",
|
|
54
|
+
"step_efficiency",
|
|
55
|
+
"safety_adherence",
|
|
56
|
+
"tool_evidence",
|
|
57
|
+
"trace_coverage",
|
|
58
|
+
"completion_judge",
|
|
59
|
+
)
|
|
60
|
+
|
|
61
|
+
# The MANDATORY deterministic post-state anchors (analogue of
|
|
62
|
+
# V1_IMAGE_LOSS_DETERMINISTIC_ANCHOR_TERMS). Both alias the SAME state_quality
|
|
63
|
+
# anchor (optimize.py:17565-17568 — url/final_state/order_id match); both admitted
|
|
64
|
+
# so authors can name the outcome either way. A browser-surface loss MUST carry
|
|
65
|
+
# >= 1 of these (unit 2). For desktop the admissible anchor narrows to
|
|
66
|
+
# V1_CUA_DESKTOP_ANCHOR_TERMS (unit 4).
|
|
67
|
+
V1_CUA_LOSS_DETERMINISTIC_ANCHOR_TERMS = ("task_success", "state_match")
|
|
68
|
+
|
|
69
|
+
# The narrower deterministic anchor admitted under cua_surface=desktop
|
|
70
|
+
# (ARCH §2.4 / 9C-A7b) — grounding/step accuracy, NOT full task success.
|
|
71
|
+
V1_CUA_DESKTOP_ANCHOR_TERMS = ("grounding_step_accuracy",)
|
|
72
|
+
|
|
73
|
+
# The bounded/guarded judge contributors (analogue of V1_IMAGE_LOSS_JUDGE_TERMS);
|
|
74
|
+
# keyed (unit 4). A judge-only loss (terms subset of this set) is structurally
|
|
75
|
+
# rejected (unit 2 / 9C-D3).
|
|
76
|
+
V1_CUA_LOSS_JUDGE_TERMS = ("completion_judge",)
|
|
77
|
+
|
|
78
|
+
# The dual-objective safety axis (ST-WebAgentBench CuP / HarmonyGuard); the
|
|
79
|
+
# unsafe-completion canary rides it (unit 2). CuP-proxy:
|
|
80
|
+
# prompt_injection_touched==0 AND blocked_action==0 (the kit's action_quality
|
|
81
|
+
# safety half, optimize.py:17552-17554).
|
|
82
|
+
V1_CUA_LOSS_MANDATORY_SAFETY_TERMS = ("safety_adherence",)
|
|
83
|
+
|
|
84
|
+
# The four-token CUA sub-attribution closed set (9C §2.3), stamped alongside the
|
|
85
|
+
# base FAILURE_LAYERS tag (analogue of V1_IMAGE_FAILURE_SUBLAYERS).
|
|
86
|
+
V1_CUA_FAILURE_SUBLAYERS = ("perception", "grounding", "action_policy", "reasoning_memory")
|
|
87
|
+
|
|
88
|
+
# The cua_surface switch on WorldSpec.spec (analogue of V1_IMAGE_TASK_MODES). ONE
|
|
89
|
+
# world loop, two surface profiles (9C-D2): browser -> world.kind=browser (full
|
|
90
|
+
# post-state); desktop -> world.kind=computer_use (grounding/step, unit 4).
|
|
91
|
+
V1_CUA_SURFACES = ("browser", "desktop")
|
|
92
|
+
|
|
93
|
+
# A MARKER field on artifact metadata — NOT a new evidence class (R5/A18; the
|
|
94
|
+
# frozen EVIDENCE_CLASSES 4-tuple live/_contract.py:18 is unchanged). The analogue
|
|
95
|
+
# of V1_IMAGE_FIDELITY_TIERS. (ARCH §2.6)
|
|
96
|
+
V1_CUA_FIDELITY_TIERS = ("deterministic_fixture", "keyed_live_model")
|
|
97
|
+
|
|
98
|
+
# The typed ``kind`` discriminators a completion guard row may carry, beyond the
|
|
99
|
+
# base sentinel/canary rows the loss guard already allows (ARCH §2.2; the analogue
|
|
100
|
+
# of V1_IMAGE_PERCEPTION_GUARD_KINDS).
|
|
101
|
+
V1_CUA_COMPLETION_GUARD_KINDS = ("fake_completion", "unsafe_completion")
|
|
102
|
+
|
|
103
|
+
# NAMING MIRROR ONLY (9C-A1c / 9C-D4). References the kit's EXISTING mutation-pack
|
|
104
|
+
# operators (normalize_browser_mutation_pack, environment.py:5146;
|
|
105
|
+
# _browser_mutation_perturbations, :28727; the prompt-injection surfaces,
|
|
106
|
+
# :29350/:2903). There is NO cua_perturb.py and NO apply_cua_perturbations
|
|
107
|
+
# function — the sharpest contrast with 9B's image_perturb.py.
|
|
108
|
+
V1_CUA_PERTURBATION_OPERATORS = ("selector_drift", "layout_shift", "stale_screenshot", "injected_dom")
|
|
109
|
+
|
|
110
|
+
# The registered world-kind tokens + the namespaced extension names (R4 hook). The
|
|
111
|
+
# CUA loop registers EXECUTABLE-LOOP status for two already-frozen kinds.
|
|
112
|
+
CUA_BROWSER_WORLD_KIND = "browser"
|
|
113
|
+
CUA_DESKTOP_WORLD_KIND = "computer_use"
|
|
114
|
+
CUA_BROWSER_EXTENSION_NAME = "agentlearning.browser_cua"
|
|
115
|
+
CUA_DESKTOP_EXTENSION_NAME = "agentlearning.computer_use_cua"
|
|
116
|
+
|
|
117
|
+
# The R4 rung -> evidence-class ladder (ARCH §2.6). The deterministic core is
|
|
118
|
+
# local_gate/captured_fixture; live_lane is added ONLY on the keyed lane record
|
|
119
|
+
# (unit 7), never the day-one deterministic record.
|
|
120
|
+
_CUA_RUNG_LADDER = {
|
|
121
|
+
"rung1": ["local_gate"],
|
|
122
|
+
"perturbed": ["live_stressed", "captured_fixture"],
|
|
123
|
+
"keyed_browser_vm": ["live_lane"],
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
class CuaLossCompositionError(ValueError):
|
|
128
|
+
"""Raised when a CUA objective violates the 9C-A2 composition rule (the
|
|
129
|
+
``cua_loss_guard_missing`` finding — a CUA specialization of
|
|
130
|
+
``objective_guards_missing``). A ``ValueError`` subclass so callers can
|
|
131
|
+
``except ValueError`` exactly as for ``ImageLossCompositionError`` /
|
|
132
|
+
``VoiceLossCompositionError``."""
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def _term_refs(objective: Mapping[str, Any]) -> list[str]:
|
|
136
|
+
"""The objective's eval refs (read from ``evals`` — the loss.py schema; also
|
|
137
|
+
tolerant of a ``terms`` alias)."""
|
|
138
|
+
rows = objective.get("evals") or objective.get("terms") or []
|
|
139
|
+
return [
|
|
140
|
+
str(term.get("eval"))
|
|
141
|
+
for term in rows
|
|
142
|
+
if isinstance(term, Mapping) and term.get("eval")
|
|
143
|
+
]
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def _admissible_anchor_terms(cua_surface: str) -> tuple[str, ...]:
|
|
147
|
+
"""The surface-admissible deterministic anchor set (the analogue of the image
|
|
148
|
+
``_admissible_anchor_terms(task_mode)``): browser -> the full post-state
|
|
149
|
+
anchors; desktop -> the narrower grounding/step anchor (unit 4)."""
|
|
150
|
+
if cua_surface == "desktop":
|
|
151
|
+
return V1_CUA_DESKTOP_ANCHOR_TERMS
|
|
152
|
+
return V1_CUA_LOSS_DETERMINISTIC_ANCHOR_TERMS
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def _admissible_term_refs(cua_surface: str) -> tuple[str, ...]:
|
|
156
|
+
"""The surface-admissible loss-term menu: browser -> V1_CUA_LOSS_TERM_REFS;
|
|
157
|
+
desktop -> the narrower grounding/step anchor + the deterministic-composition
|
|
158
|
+
+ safety + judge terms (unit 4). Desktop drops the browser-only post-state
|
|
159
|
+
anchors (task_success / state_match) since the credential-free desktop rung is
|
|
160
|
+
grounding/step ONLY, not full task success."""
|
|
161
|
+
if cua_surface == "desktop":
|
|
162
|
+
return (
|
|
163
|
+
V1_CUA_DESKTOP_ANCHOR_TERMS
|
|
164
|
+
+ (
|
|
165
|
+
"grounding_mutation_resilience",
|
|
166
|
+
"action_correctness",
|
|
167
|
+
"step_efficiency",
|
|
168
|
+
"safety_adherence",
|
|
169
|
+
"tool_evidence",
|
|
170
|
+
"trace_coverage",
|
|
171
|
+
)
|
|
172
|
+
+ V1_CUA_LOSS_JUDGE_TERMS
|
|
173
|
+
)
|
|
174
|
+
return V1_CUA_LOSS_TERM_REFS
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def attribute_cua_sublayer(
|
|
178
|
+
*,
|
|
179
|
+
failure_layer: str,
|
|
180
|
+
deficit: Mapping[str, Any] | None = None,
|
|
181
|
+
signal: str | None = None,
|
|
182
|
+
) -> str:
|
|
183
|
+
"""Map a weak CUA cell to a ``V1_CUA_FAILURE_SUBLAYERS`` token, stamped
|
|
184
|
+
ALONGSIDE the base ``FAILURE_LAYERS`` tag (a weak cell carries both, e.g.
|
|
185
|
+
``{failure_layer:"agent_behavior", cua_sublayer:"grounding"}``). The base
|
|
186
|
+
attribution rides the existing ``practice/_diagnose.py`` machinery; this is the
|
|
187
|
+
thin sublayer helper (the CUA analogue of ``attribute_image_sublayer``,
|
|
188
|
+
image_loop.py:208).
|
|
189
|
+
|
|
190
|
+
Routing (ARCH-9C §2.3 table, grounded in the observe->ground->act decomposition
|
|
191
|
+
+ the kit's layers ["browser","cua","security","evaluator"] optimize.py:17267 +
|
|
192
|
+
the step-level stuck/milestone split 2604.27151):
|
|
193
|
+
* stale screenshot, didn't refresh; missed an observed change -> ``perception``
|
|
194
|
+
(observation-channel failure);
|
|
195
|
+
* selector drifted, mis-clicked; coordinate off -> ``grounding`` (the
|
|
196
|
+
observe->ground seam, the dominant mutation-resilience failure);
|
|
197
|
+
* looped on the same step / 2.5-2.8x too many steps; touched injected banner
|
|
198
|
+
-> ``action_policy`` (action/escalation policy + safety);
|
|
199
|
+
* right perception, wrong plan; bad memory of prior steps ->
|
|
200
|
+
``reasoning_memory`` (plan/memory failure — ACuRL / Reflexion)."""
|
|
201
|
+
|
|
202
|
+
sig = str(signal or (deficit or {}).get("signal") or "").lower()
|
|
203
|
+
# Precedence-ordered (the ARCH §2.3 routing table). reasoning_memory is
|
|
204
|
+
# checked FIRST among the higher-cognition cues so a "right perception, wrong
|
|
205
|
+
# plan; bad memory of prior steps" cell routes to reasoning_memory even though
|
|
206
|
+
# it mentions perception (the word is a red herring; the ARCH perception row is
|
|
207
|
+
# "stale screenshot / missed an observed change", which carries none of these
|
|
208
|
+
# cues).
|
|
209
|
+
if any(
|
|
210
|
+
k in sig
|
|
211
|
+
for k in ("wrong plan", "wrong_plan", "bad memory", "memory", "reflect",
|
|
212
|
+
"reasoning", "prior step", "prior_step")
|
|
213
|
+
):
|
|
214
|
+
return "reasoning_memory"
|
|
215
|
+
if any(
|
|
216
|
+
k in sig
|
|
217
|
+
for k in ("stale screenshot", "stale_screenshot", "didn't refresh",
|
|
218
|
+
"did not refresh", "missed an observed", "missed_change",
|
|
219
|
+
"observed change", "screenshot")
|
|
220
|
+
):
|
|
221
|
+
return "perception"
|
|
222
|
+
if any(
|
|
223
|
+
k in sig
|
|
224
|
+
for k in ("selector drift", "selector_drift", "drifted selector",
|
|
225
|
+
"selector drifted", "mis-click", "misclick", "mis click",
|
|
226
|
+
"coordinate off", "coordinate_off", "ground")
|
|
227
|
+
):
|
|
228
|
+
return "grounding"
|
|
229
|
+
if any(
|
|
230
|
+
k in sig
|
|
231
|
+
for k in ("loop", "too many steps", "step_efficiency", "redundant",
|
|
232
|
+
"injected banner", "injected_banner", "injection", "blocked",
|
|
233
|
+
"escalation", "action_policy", "action policy", "unsafe")
|
|
234
|
+
):
|
|
235
|
+
return "action_policy"
|
|
236
|
+
# default: infra-implicated cells land on perception (the cheapest observation
|
|
237
|
+
# fix before blaming the policy); otherwise the reasoning/memory layer.
|
|
238
|
+
if failure_layer in ("lane_infra", "framework_runtime", "provider"):
|
|
239
|
+
return "perception"
|
|
240
|
+
return "reasoning_memory"
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
def compile_cua_objective(
|
|
244
|
+
payload: Mapping[str, Any], *, cua_surface: str = "browser"
|
|
245
|
+
) -> dict:
|
|
246
|
+
"""Compile a multi-objective CUA loss with a fake/unsafe-completion Goodhart
|
|
247
|
+
guard (ARCH-9C §2.2 / 9C-A2 / 9C-D3). The CUA analogue of
|
|
248
|
+
``compile_image_objective`` (image_loop.py:132). Enforces, ON TOP of the
|
|
249
|
+
verbatim ``loss.compile_objective`` Goodhart guard:
|
|
250
|
+
|
|
251
|
+
rule 1: closed-set ``cua_surface`` (browser | desktop);
|
|
252
|
+
rule 2: >= 2 terms (a single-term CUA objective is reward-hackable);
|
|
253
|
+
rule 3: >= 1 surface-admissible deterministic post-state anchor — a
|
|
254
|
+
judge-only loss is INVALID (9C-D3). ``cua_surface`` selects the
|
|
255
|
+
admissible anchor set: browser -> V1_CUA_LOSS_DETERMINISTIC_ANCHOR_TERMS;
|
|
256
|
+
desktop -> V1_CUA_DESKTOP_ANCHOR_TERMS (the narrower grounding/step
|
|
257
|
+
anchor, unit 4);
|
|
258
|
+
rule 4: unknown-ref rejection (every term must be a member of the surface
|
|
259
|
+
menu);
|
|
260
|
+
rule 5: when sentinel/canary rows carry a completion ``kind`` discriminator
|
|
261
|
+
it must be in V1_CUA_COMPLETION_GUARD_KINDS (the closed set).
|
|
262
|
+
|
|
263
|
+
Then delegates to ``loss.compile_objective`` VERBATIM — which unconditionally
|
|
264
|
+
enforces the populated guard block (sentinel_rows / canary_evals,
|
|
265
|
+
min_guard_count >= 1, "There is no override.")."""
|
|
266
|
+
|
|
267
|
+
from . import loss as _loss # downward facade import (legal; image_loop.py idiom)
|
|
268
|
+
|
|
269
|
+
# rule 1: closed-set cua_surface.
|
|
270
|
+
if cua_surface not in V1_CUA_SURFACES:
|
|
271
|
+
raise CuaLossCompositionError(
|
|
272
|
+
f"cua_loss_guard_missing: cua_surface {cua_surface!r} not in "
|
|
273
|
+
f"{V1_CUA_SURFACES}"
|
|
274
|
+
)
|
|
275
|
+
|
|
276
|
+
refs = _term_refs(payload)
|
|
277
|
+
|
|
278
|
+
# rule 2: >= 2 terms.
|
|
279
|
+
if len(refs) < 2:
|
|
280
|
+
raise CuaLossCompositionError(
|
|
281
|
+
"cua_loss_guard_missing: a CUA objective is reward-hackable as a "
|
|
282
|
+
"single term; it MUST be multi-objective (>= 2 terms). "
|
|
283
|
+
f"got {refs}"
|
|
284
|
+
)
|
|
285
|
+
|
|
286
|
+
# rule 3: >= 1 surface-admissible deterministic post-state anchor (judge-only
|
|
287
|
+
# REJECTED — 9C-D3).
|
|
288
|
+
anchors = _admissible_anchor_terms(cua_surface)
|
|
289
|
+
if not any(ref in anchors for ref in refs):
|
|
290
|
+
raise CuaLossCompositionError(
|
|
291
|
+
"cua_loss_guard_missing: a CUA loss MUST carry >= 1 deterministic "
|
|
292
|
+
f"post-state anchor {anchors}; a judge-only loss is INVALID by "
|
|
293
|
+
f"contract (9C-D3). got {refs} (cua_surface={cua_surface})"
|
|
294
|
+
)
|
|
295
|
+
|
|
296
|
+
# rule 4: unknown-ref rejection (surface-filtered).
|
|
297
|
+
allowed = _admissible_term_refs(cua_surface)
|
|
298
|
+
for ref in refs:
|
|
299
|
+
if ref not in allowed:
|
|
300
|
+
raise CuaLossCompositionError(
|
|
301
|
+
f"cua_loss_guard_missing: unknown CUA loss term {ref!r}; "
|
|
302
|
+
f"expected members of {allowed} (cua_surface={cua_surface})"
|
|
303
|
+
)
|
|
304
|
+
|
|
305
|
+
# rule 5: the fake/unsafe-completion guard rows ride the existing
|
|
306
|
+
# sentinel_rows/canary_evals with a typed ``kind`` discriminator (no new
|
|
307
|
+
# ObjectiveSpec field, ARCH-9C §2.2). When present it must be in the closed set
|
|
308
|
+
# (plus any untyped/base rows the loss guard already allows).
|
|
309
|
+
guards = payload.get("guards") or {}
|
|
310
|
+
for bucket in ("sentinel_rows", "canary_evals"):
|
|
311
|
+
for row in guards.get(bucket) or []:
|
|
312
|
+
if isinstance(row, Mapping):
|
|
313
|
+
kind = row.get("kind")
|
|
314
|
+
if kind is not None and kind not in V1_CUA_COMPLETION_GUARD_KINDS:
|
|
315
|
+
raise CuaLossCompositionError(
|
|
316
|
+
f"cua_loss_guard_missing: guard row kind {kind!r} not in "
|
|
317
|
+
f"{V1_CUA_COMPLETION_GUARD_KINDS}"
|
|
318
|
+
)
|
|
319
|
+
|
|
320
|
+
# the verbatim Goodhart guard (loss.py:106-116) — "There is no override."
|
|
321
|
+
return _loss.compile_objective(payload)
|
|
322
|
+
|
|
323
|
+
|
|
324
|
+
def _validate_cua_world_spec(spec: Mapping[str, Any]) -> None:
|
|
325
|
+
"""The R4 ``spec_validator`` for the CUA world: validate the ``cua_surface``
|
|
326
|
+
switch (browser | desktop) on ``WorldSpec.spec`` (the analogue of
|
|
327
|
+
``_validate_image_world_spec``, image_loop.py:293). Raises ValueError on an
|
|
328
|
+
unknown surface (the closed-set guard)."""
|
|
329
|
+
|
|
330
|
+
cua_surface = str((spec or {}).get("cua_surface", "browser"))
|
|
331
|
+
if cua_surface not in V1_CUA_SURFACES:
|
|
332
|
+
raise ValueError(
|
|
333
|
+
f"cua world.spec.cua_surface {cua_surface!r} not in {V1_CUA_SURFACES}"
|
|
334
|
+
)
|
|
335
|
+
|
|
336
|
+
|
|
337
|
+
def _ensure_cua_world_registered(cua_surface: str = "browser") -> None:
|
|
338
|
+
"""Flip ``browser`` / ``computer_use`` from typed-only to EXECUTABLE-LOOP
|
|
339
|
+
status via the R4 hook (ARCH-9C §2.1 / §2.3 / 9C-D2 / 9C-A1b). Idempotent BY
|
|
340
|
+
VENDOR.NAME.
|
|
341
|
+
|
|
342
|
+
THE 9C-A1b RULE (binding): this does NOT use the verbatim image idempotence
|
|
343
|
+
guard (``if kind in resolved_world_kinds(): return``, image_loop.py:272),
|
|
344
|
+
because ``browser`` / ``computer_use`` are ALREADY in ``resolved_world_kinds()``
|
|
345
|
+
as frozen built-ins (contract.py:55) — that guard would short-circuit
|
|
346
|
+
IMMEDIATELY and never record the R4 executable-loop evidence (spec_validator +
|
|
347
|
+
rung_ladder + rung1_fixture_green). Instead it gates on the EXECUTABLE-LOOP
|
|
348
|
+
RECORD presence in ``_EXTRA_WORLD_KINDS`` (the additive R4 record keyed by
|
|
349
|
+
vendor.name). ``register_world_kind`` pushes the CUA record into
|
|
350
|
+
``_EXTRA_WORLD_KINDS``; built-ins shadow extensions at resolution
|
|
351
|
+
(contract.py:88-91), so ``WorldSpec(kind="browser")`` keeps validating against
|
|
352
|
+
the built-in entry and the frozen ``SIMULATION_WORLD_KINDS`` tuple stays
|
|
353
|
+
byte-stable. ``browser`` / ``computer_use`` stay in
|
|
354
|
+
``TYPED_ONLY_WORLD_KINDS_V1`` — executable-loop status is carried by the
|
|
355
|
+
registry record + the ``cua_loop_readiness`` gate, NOT by the frozen
|
|
356
|
+
executable tuple."""
|
|
357
|
+
|
|
358
|
+
from fi.simulate.simulation import contract as _contract
|
|
359
|
+
|
|
360
|
+
if cua_surface not in V1_CUA_SURFACES:
|
|
361
|
+
raise CuaLossCompositionError(
|
|
362
|
+
f"cua_loss_guard_missing: cua_surface {cua_surface!r} not in "
|
|
363
|
+
f"{V1_CUA_SURFACES}"
|
|
364
|
+
)
|
|
365
|
+
|
|
366
|
+
if cua_surface == "browser":
|
|
367
|
+
kind_token = CUA_BROWSER_WORLD_KIND
|
|
368
|
+
vendor_name = CUA_BROWSER_EXTENSION_NAME
|
|
369
|
+
else:
|
|
370
|
+
kind_token = CUA_DESKTOP_WORLD_KIND
|
|
371
|
+
vendor_name = CUA_DESKTOP_EXTENSION_NAME
|
|
372
|
+
|
|
373
|
+
from . import extensions as _ext
|
|
374
|
+
|
|
375
|
+
# 9C-A1b: gate on the EXECUTABLE-LOOP RECORD, not on bare admissibility. The
|
|
376
|
+
# kind is already admissible (built-in); the executable-loop marker is the
|
|
377
|
+
# _EXTRA_WORLD_KINDS record carrying the CUA kind_token + the vendor.name.
|
|
378
|
+
# register_extension keys _EXTRA_WORLD_KINDS by the kind_token (it calls
|
|
379
|
+
# contract.register_world_kind(token, stored)), so the contract record is read
|
|
380
|
+
# by kind_token; the vendor.name lives inside the record's ``name`` field
|
|
381
|
+
# (the idempotence key per 9C-A1b — one executable-loop record per vendor).
|
|
382
|
+
existing = _contract._EXTRA_WORLD_KINDS.get(kind_token) # additive record, never the built-in
|
|
383
|
+
if existing and existing.get("kind_token") == kind_token and existing.get("name") == vendor_name:
|
|
384
|
+
return # executable-loop record already present (idempotent by vendor.name)
|
|
385
|
+
|
|
386
|
+
# The persistent extension registry (extensions._REGISTRY) outlives the
|
|
387
|
+
# contract's _EXTRA_WORLD_KINDS within a process. register_extension RAISES on
|
|
388
|
+
# a name collision, so if the extension is ALREADY in the registry (e.g. the
|
|
389
|
+
# contract record was cleared but the registry was not), re-push the existing
|
|
390
|
+
# stored record into the contract directly rather than re-registering. This
|
|
391
|
+
# keeps the gate idempotent by vendor.name AND restores the executable-loop
|
|
392
|
+
# evidence in _EXTRA_WORLD_KINDS.
|
|
393
|
+
stored = _ext.resolve("environment", vendor_name)
|
|
394
|
+
if stored is not None and stored.get("kind_token") == kind_token:
|
|
395
|
+
_contract.register_world_kind(kind_token, stored)
|
|
396
|
+
return
|
|
397
|
+
|
|
398
|
+
_ext.register_extension(
|
|
399
|
+
"environment",
|
|
400
|
+
{
|
|
401
|
+
"name": vendor_name, # vendor.name shape (_validate_record)
|
|
402
|
+
"kind_token": kind_token, # "browser" / "computer_use" token
|
|
403
|
+
"spec_validator": _validate_cua_world_spec, # R4 mandate
|
|
404
|
+
"rung_ladder": _CUA_RUNG_LADDER, # R4 mandate
|
|
405
|
+
# the deterministic core; live_lane is added ONLY on the keyed lane
|
|
406
|
+
# record (unit 7), never here. gated_contexts_runnable stays False
|
|
407
|
+
# until rung1_fixture_green (extensions.py admission); 9C never
|
|
408
|
+
# silently claims executable.
|
|
409
|
+
"evidence_class_capability": ["local_gate", "captured_fixture"],
|
|
410
|
+
},
|
|
411
|
+
)
|
|
412
|
+
|
|
413
|
+
|
|
414
|
+
def build_cua_practice_loop_manifest(
|
|
415
|
+
*,
|
|
416
|
+
name: str,
|
|
417
|
+
base_agent: Mapping[str, Any],
|
|
418
|
+
search_space: Mapping[str, Sequence[Any]],
|
|
419
|
+
objective: Mapping[str, Any],
|
|
420
|
+
eval_budget: int,
|
|
421
|
+
seed: int,
|
|
422
|
+
cua_surface: str = "browser",
|
|
423
|
+
scenario_inline: Optional[Mapping[str, Any]] = None,
|
|
424
|
+
max_rounds: int = 8,
|
|
425
|
+
) -> dict[str, Any]:
|
|
426
|
+
"""Assemble the CUA improvement-loop manifest: the 13D Practice Loop on
|
|
427
|
+
``world.kind=browser`` / ``computer_use`` + ``cua_surface`` with the
|
|
428
|
+
multi-objective guarded CUA loss + the whole CUA-agent search space (9C-D5).
|
|
429
|
+
Delegates to ``optimize.build_practice_loop_manifest`` so its validators hold
|
|
430
|
+
VERBATIM (9C-A3). The objective is compiled by ``compile_cua_objective`` (the
|
|
431
|
+
9C-A2 rule) before it rides the simulation.
|
|
432
|
+
|
|
433
|
+
Byte-parallel to ``build_image_practice_loop_manifest`` except: (a) the
|
|
434
|
+
``_ensure_cua_world_registered(cua_surface)`` call uses the 9C-A1b
|
|
435
|
+
executable-loop-record gate (NOT the verbatim image idempotence guard); (b)
|
|
436
|
+
``world["kind"]`` is ``"browser"`` / ``"computer_use"`` driven by
|
|
437
|
+
``cua_surface`` (image's is always ``"image"``); (c) the ``spec["cua_surface"]``
|
|
438
|
+
write (instead of ``spec["task_mode"]``)."""
|
|
439
|
+
|
|
440
|
+
from . import optimize as _optimize # downward facade import (legal)
|
|
441
|
+
|
|
442
|
+
if cua_surface not in V1_CUA_SURFACES:
|
|
443
|
+
raise CuaLossCompositionError(
|
|
444
|
+
f"cua_loss_guard_missing: cua_surface {cua_surface!r} not in "
|
|
445
|
+
f"{V1_CUA_SURFACES}"
|
|
446
|
+
)
|
|
447
|
+
|
|
448
|
+
_ensure_cua_world_registered(cua_surface) # step 1 (§3.1, 9C-A1b)
|
|
449
|
+
compiled = compile_cua_objective(objective, cua_surface=cua_surface) # step 2 (unit 2)
|
|
450
|
+
inline = dict(scenario_inline or {})
|
|
451
|
+
inline.setdefault("version", "agent-learning.simulation.v1")
|
|
452
|
+
inline["objective"] = compiled
|
|
453
|
+
world = dict(inline.get("world") or {})
|
|
454
|
+
world["kind"] = (
|
|
455
|
+
CUA_BROWSER_WORLD_KIND if cua_surface == "browser" else CUA_DESKTOP_WORLD_KIND
|
|
456
|
+
) # step 3 — the registered kind
|
|
457
|
+
spec = dict(world.get("spec") or {})
|
|
458
|
+
spec["cua_surface"] = cua_surface # the cua_surface switch on WorldSpec.spec
|
|
459
|
+
world["spec"] = spec
|
|
460
|
+
inline["world"] = world
|
|
461
|
+
|
|
462
|
+
return _optimize.build_practice_loop_manifest( # step 4 — VERBATIM delegate
|
|
463
|
+
name=name,
|
|
464
|
+
simulation={"version": inline["version"], "inline": inline},
|
|
465
|
+
base_agent=base_agent,
|
|
466
|
+
search_space=search_space,
|
|
467
|
+
eval_budget=eval_budget,
|
|
468
|
+
seed=seed,
|
|
469
|
+
max_rounds=max_rounds,
|
|
470
|
+
)
|
|
471
|
+
|
|
472
|
+
|
|
473
|
+
# === Unit 7 — the keyed real-browser/VM lane (opt-in, NEVER a gate prerequisite) ===
|
|
474
|
+
# ARCH-9C §2.4 / §2.6 / 9C-D1/D6/A8. The keyed completion_judge term, the desktop
|
|
475
|
+
# full-post-state rungs, and the one real-browser/CUA-agent live-proof are
|
|
476
|
+
# owner-keyed/infra-provisioned, opt-in, never a release gate. The deterministic
|
|
477
|
+
# core stays local_gate/captured_fixture; the keyed lane is the ONLY honest place
|
|
478
|
+
# for live_lane (a real keyed browser/VM/judge ran).
|
|
479
|
+
|
|
480
|
+
KEYED_CUA_EXTENSION_NAME = "agentlearning.cua.keyed"
|
|
481
|
+
|
|
482
|
+
# the env keys that gate the keyed completion_judge lane (checked, never required
|
|
483
|
+
# by any gate). The GRADE-shaped judge term (9C-A8) calls a judge model.
|
|
484
|
+
CUA_JUDGE_KEY_ENVS = ("AGENT_LEARNING_CUA_JUDGE_KEY", "OPENAI_API_KEY")
|
|
485
|
+
|
|
486
|
+
|
|
487
|
+
class CuaKeyedLaneUnavailable(RuntimeError):
|
|
488
|
+
"""Raised by the keyed lane when no judge key / VM infra is present — the loud
|
|
489
|
+
refusal (the ``cua_judge_key_unavailable`` / ``cua_desktop_infra_unavailable``
|
|
490
|
+
finding). The deterministic core NEVER raises this; only the opt-in keyed path
|
|
491
|
+
does."""
|
|
492
|
+
|
|
493
|
+
|
|
494
|
+
def cua_judge_key_present() -> bool:
|
|
495
|
+
"""True iff a judge key is configured for the keyed completion_judge lane."""
|
|
496
|
+
import os
|
|
497
|
+
|
|
498
|
+
return any(os.environ.get(env) for env in CUA_JUDGE_KEY_ENVS)
|
|
499
|
+
|
|
500
|
+
|
|
501
|
+
def register_keyed_cua_lane() -> None:
|
|
502
|
+
"""Register the SEPARATE keyed-lane extension record that adds ``live_lane`` to
|
|
503
|
+
``evidence_class_capability`` (ARCH-9C §2.6, unit 7). Idempotent. This is the
|
|
504
|
+
ONLY record that may carry ``live_lane`` — the deterministic-core records
|
|
505
|
+
(``_ensure_cua_world_registered``) stay ``("local_gate","captured_fixture")``.
|
|
506
|
+
NEVER called by the gate; opt-in only."""
|
|
507
|
+
from . import extensions as _ext
|
|
508
|
+
|
|
509
|
+
if _ext.resolve("environment", KEYED_CUA_EXTENSION_NAME) is not None:
|
|
510
|
+
return
|
|
511
|
+
_ext.register_extension(
|
|
512
|
+
"environment",
|
|
513
|
+
{
|
|
514
|
+
"name": KEYED_CUA_EXTENSION_NAME,
|
|
515
|
+
# the keyed lane declares the keyed capability without a second
|
|
516
|
+
# kind_token (the world kinds are already registered).
|
|
517
|
+
"evidence_class_capability": ["local_gate", "captured_fixture", "live_lane"],
|
|
518
|
+
},
|
|
519
|
+
)
|
|
520
|
+
|
|
521
|
+
|
|
522
|
+
def run_keyed_cua_live_proof(
|
|
523
|
+
*,
|
|
524
|
+
base_agent: Mapping[str, Any],
|
|
525
|
+
search_space: Mapping[str, Sequence[Any]],
|
|
526
|
+
objective: Mapping[str, Any],
|
|
527
|
+
eval_budget: int,
|
|
528
|
+
seed: int,
|
|
529
|
+
cua_surface: str = "browser",
|
|
530
|
+
name: str = "cua-keyed-live-proof",
|
|
531
|
+
) -> dict[str, Any]:
|
|
532
|
+
"""The one owner-keyed live-proof entry (WORKFLOW Step 5 real-keys ground
|
|
533
|
+
rule). Refuses LOUDLY without a key (``CuaKeyedLaneUnavailable`` ->
|
|
534
|
+
``cua_judge_key_unavailable``) — never a fake number, never a release
|
|
535
|
+
prerequisite. With a key, it builds the manifest and marks the run
|
|
536
|
+
``live_lane`` / ``fidelity_tier=keyed_live_model``.
|
|
537
|
+
|
|
538
|
+
The keyed run itself (calling the GRADE judge / a real browser / a VM) is left
|
|
539
|
+
to the caller's runtime; this returns the keyed manifest + the honest
|
|
540
|
+
evidence-class stamp so an owner can execute it once with real keys/infra."""
|
|
541
|
+
if not cua_judge_key_present():
|
|
542
|
+
raise CuaKeyedLaneUnavailable(
|
|
543
|
+
"cua_judge_key_unavailable: the keyed real-browser/VM lane requires a "
|
|
544
|
+
f"judge key (one of {CUA_JUDGE_KEY_ENVS}); withheld -- never a fake "
|
|
545
|
+
"number, never a release prerequisite"
|
|
546
|
+
)
|
|
547
|
+
register_keyed_cua_lane()
|
|
548
|
+
manifest = build_cua_practice_loop_manifest(
|
|
549
|
+
name=name,
|
|
550
|
+
base_agent=base_agent,
|
|
551
|
+
search_space=search_space,
|
|
552
|
+
objective=objective,
|
|
553
|
+
eval_budget=eval_budget,
|
|
554
|
+
seed=seed,
|
|
555
|
+
cua_surface=cua_surface,
|
|
556
|
+
)
|
|
557
|
+
return {
|
|
558
|
+
"manifest": manifest,
|
|
559
|
+
"evidence_class": "live_lane", # the ONLY honest live_lane
|
|
560
|
+
"fidelity_tier": "keyed_live_model",
|
|
561
|
+
"cua_surface": cua_surface,
|
|
562
|
+
}
|