agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
fi/alk/extensions.py
ADDED
|
@@ -0,0 +1,163 @@
|
|
|
1
|
+
"""Unit 4 (BBG U4 / ARCH §2e) — the four 13D-4 registries + extension_admission.
|
|
2
|
+
|
|
3
|
+
Explicit in-process registration (AD-J: no entry-points, no import-time
|
|
4
|
+
discovery — local-first, gate-scannable). One uniform record shape across four
|
|
5
|
+
points; ONE choke point (``extension_admission``) enforces 13D-D5 (extension ≠
|
|
6
|
+
exemption) in gated contexts. The facade pushes world-kind/role registrations
|
|
7
|
+
DOWN into ``fi.simulate.simulation.contract`` setters (Appendix C-1); the engine
|
|
8
|
+
never imports up.
|
|
9
|
+
"""
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from typing import Any, Dict, Mapping, Optional
|
|
13
|
+
|
|
14
|
+
from .live._contract import EVIDENCE_CLASSES, RELEASE_ADMISSIBLE_EVIDENCE_CLASSES
|
|
15
|
+
|
|
16
|
+
EXTENSION_POINTS = ("environment", "loss", "optimizer", "generator")
|
|
17
|
+
|
|
18
|
+
# point -> name -> record
|
|
19
|
+
_REGISTRY: Dict[str, Dict[str, dict]] = {point: {} for point in EXTENSION_POINTS}
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class ExtensionError(ValueError):
|
|
23
|
+
"""Raised when a registration record is malformed."""
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def _validate_record(point: str, record: Mapping[str, Any]) -> dict:
|
|
27
|
+
if point not in EXTENSION_POINTS:
|
|
28
|
+
raise ExtensionError(f"point {point!r} not in {EXTENSION_POINTS}")
|
|
29
|
+
name = record.get("name")
|
|
30
|
+
if not name or "." not in str(name):
|
|
31
|
+
raise ExtensionError(
|
|
32
|
+
"extension record requires a namespaced name 'vendor.name'"
|
|
33
|
+
)
|
|
34
|
+
if str(name) in _REGISTRY[point]:
|
|
35
|
+
raise ExtensionError(f"extension name collision: {name!r} already registered for {point}")
|
|
36
|
+
caps = record.get("evidence_class_capability") or []
|
|
37
|
+
for cap in caps:
|
|
38
|
+
if cap not in EVIDENCE_CLASSES:
|
|
39
|
+
raise ExtensionError(
|
|
40
|
+
f"evidence_class_capability {cap!r} not in {EVIDENCE_CLASSES}"
|
|
41
|
+
)
|
|
42
|
+
if point == "optimizer" and not record.get("declared_budgets"):
|
|
43
|
+
raise ExtensionError(
|
|
44
|
+
"optimizer extensions REQUIRE declared_budgets (refused otherwise)"
|
|
45
|
+
)
|
|
46
|
+
stored = dict(record)
|
|
47
|
+
stored["point"] = point
|
|
48
|
+
stored.setdefault("provides", None)
|
|
49
|
+
stored.setdefault("conformance_manifest", None)
|
|
50
|
+
stored.setdefault("evidence_class_capability", list(caps))
|
|
51
|
+
stored.setdefault("version", "0.0.0")
|
|
52
|
+
stored["gated_contexts_runnable"] = False # until a green conformance run lands
|
|
53
|
+
return stored
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def register_extension(point: str, record: Mapping[str, Any]) -> dict:
|
|
57
|
+
stored = _validate_record(point, record)
|
|
58
|
+
# world-kind/role registrations push down into the engine contract (C-1).
|
|
59
|
+
if point == "environment":
|
|
60
|
+
token = stored.get("kind_token")
|
|
61
|
+
if token:
|
|
62
|
+
if not stored.get("spec_validator") or not stored.get("rung_ladder"):
|
|
63
|
+
raise ExtensionError(
|
|
64
|
+
"a custom world.kind extension MUST declare spec_validator + rung_ladder (R4)"
|
|
65
|
+
)
|
|
66
|
+
from fi.simulate.simulation import contract as _contract
|
|
67
|
+
_contract.register_world_kind(str(token), stored)
|
|
68
|
+
_REGISTRY[point][str(stored["name"])] = stored
|
|
69
|
+
return stored
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def register_environment(record: Mapping[str, Any]) -> dict:
|
|
73
|
+
"""Register a descriptive environment **metadata record** (studio extension).
|
|
74
|
+
|
|
75
|
+
Canon correspondence (assessment §8 Gap B): the runtime sibling
|
|
76
|
+
``fi.simulate.registry.register_environment`` registers a **runnable** plugin
|
|
77
|
+
factory in ``environment_registry``. This one records metadata (and, for a
|
|
78
|
+
``world.kind`` extension carrying a ``kind_token``, writes the contract's
|
|
79
|
+
extension side-table via ``contract.register_world_kind`` — the frozen canon
|
|
80
|
+
constants never mutate). A record is not a factory — the two are deliberately
|
|
81
|
+
unwired.
|
|
82
|
+
"""
|
|
83
|
+
return register_extension("environment", record)
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def register_objective(record: Mapping[str, Any]) -> dict:
|
|
87
|
+
return register_extension("loss", record)
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def register_optimizer(record: Mapping[str, Any]) -> dict:
|
|
91
|
+
return register_extension("optimizer", record)
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def register_generator(record: Mapping[str, Any]) -> dict:
|
|
95
|
+
return register_extension("generator", record)
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def register_role(record: Mapping[str, Any]) -> dict:
|
|
99
|
+
"""Register a namespaced cast role (pushes into the engine contract)."""
|
|
100
|
+
stored = _validate_record("environment", {**record, "name": record["name"]})
|
|
101
|
+
token = stored.get("kind_token") or stored.get("role")
|
|
102
|
+
from fi.simulate.simulation import contract as _contract
|
|
103
|
+
if token:
|
|
104
|
+
_contract.register_cast_role(str(token), stored)
|
|
105
|
+
_REGISTRY["environment"][str(stored["name"])] = stored
|
|
106
|
+
return stored
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def resolve(point: str, token: str) -> Optional[dict]:
|
|
110
|
+
"""Built-ins first — callers check the canon BEFORE consulting this."""
|
|
111
|
+
return _REGISTRY.get(point, {}).get(str(token))
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def registered(point: str) -> tuple[str, ...]:
|
|
115
|
+
return tuple(sorted(_REGISTRY.get(point, {})))
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def extension_admission(record: Mapping[str, Any], context: Mapping[str, Any]) -> dict:
|
|
119
|
+
"""THE one choke point. In gated contexts (release-check/promotion/training)
|
|
120
|
+
a registered extension that cannot produce admissible evidence does not run.
|
|
121
|
+
Returns ``{admitted: bool}`` or the structured refusal finding."""
|
|
122
|
+
gated = bool(context.get("gated"))
|
|
123
|
+
if not gated:
|
|
124
|
+
return {"admitted": True, "reason": "non_gated_passthrough"}
|
|
125
|
+
|
|
126
|
+
point = record.get("point")
|
|
127
|
+
caps = set(record.get("evidence_class_capability") or [])
|
|
128
|
+
admissible_caps = caps & set(RELEASE_ADMISSIBLE_EVIDENCE_CLASSES)
|
|
129
|
+
conformance_green = bool(record.get("conformance_green") or record.get("gated_contexts_runnable"))
|
|
130
|
+
|
|
131
|
+
def refuse(reason: str) -> dict:
|
|
132
|
+
return {
|
|
133
|
+
"admitted": False,
|
|
134
|
+
"type": "extension_evidence_inadmissible",
|
|
135
|
+
"level": "error",
|
|
136
|
+
"name": record.get("name"),
|
|
137
|
+
"point": point,
|
|
138
|
+
"reason": reason,
|
|
139
|
+
"remediation": (
|
|
140
|
+
"run the extension's conformance_manifest green with evidence in "
|
|
141
|
+
f"{RELEASE_ADMISSIBLE_EVIDENCE_CLASSES} before any gated use"
|
|
142
|
+
),
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
# (i) conformance manifest ran green with release-admissible evidence
|
|
146
|
+
if not conformance_green or not admissible_caps:
|
|
147
|
+
return refuse("no green conformance run with release-admissible evidence")
|
|
148
|
+
# (iii) optimizer extensions without declared_budgets never run
|
|
149
|
+
if point == "optimizer" and not record.get("declared_budgets"):
|
|
150
|
+
return refuse("optimizer extension has no declared_budgets")
|
|
151
|
+
# (iv) environment/world-kind extensions claiming an executable kind need a
|
|
152
|
+
# green rung-1 fixture run recorded.
|
|
153
|
+
if point == "environment" and record.get("kind_token"):
|
|
154
|
+
if not record.get("rung1_fixture_green"):
|
|
155
|
+
return refuse("world-kind extension lacks a green rung-1 fixture run")
|
|
156
|
+
return {"admitted": True, "reason": "gated_admitted"}
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def _reset_extensions() -> None: # test-only
|
|
160
|
+
for point in EXTENSION_POINTS:
|
|
161
|
+
_REGISTRY[point].clear()
|
|
162
|
+
from fi.simulate.simulation import contract as _contract
|
|
163
|
+
_contract._reset_contract_extensions()
|
|
@@ -0,0 +1,231 @@
|
|
|
1
|
+
# ALK harness execution architecture
|
|
2
|
+
|
|
3
|
+
Implementation and test evidence are tracked in
|
|
4
|
+
[`IMPLEMENTATION_AND_VALIDATION_STATUS.md`](IMPLEMENTATION_AND_VALIDATION_STATUS.md). Repository
|
|
5
|
+
packaging behavior and the conformance matrix are documented in
|
|
6
|
+
[`ENVIRONMENT_CONFORMANCE.md`](ENVIRONMENT_CONFORMANCE.md).
|
|
7
|
+
|
|
8
|
+
## Ownership
|
|
9
|
+
|
|
10
|
+
ALK owns all execution behavior. The Future AGI platform is a control and data plane only.
|
|
11
|
+
|
|
12
|
+
| Concern | ALK package | Platform | Hosted sandbox fleet |
|
|
13
|
+
|---|---:|---:|---:|
|
|
14
|
+
| Understand repository and agent | yes | no | runs ALK |
|
|
15
|
+
| Build databases, tools, mocks and seed data | yes | no | runs ALK |
|
|
16
|
+
| Generate/validate scenarios and personas | yes | no | runs ALK |
|
|
17
|
+
| Connect to and simulate the agent | yes | no | runs ALK |
|
|
18
|
+
| Grade evidence and create artifacts | yes | no | runs ALK |
|
|
19
|
+
| Repository/secret authorization | consumes references | yes | resolves job-scoped values |
|
|
20
|
+
| Job UI, chat, cancellation and history | no | yes | reports status |
|
|
21
|
+
| Event/result/artifact storage | emits data | yes | forwards data |
|
|
22
|
+
|
|
23
|
+
No harness stage imports Temporal, Django, platform models, or platform worker code.
|
|
24
|
+
|
|
25
|
+
## One engine, two deployments
|
|
26
|
+
|
|
27
|
+
```text
|
|
28
|
+
Local CLI Hosted product
|
|
29
|
+
───────── ──────────────
|
|
30
|
+
agent-learn harness auto platform creates HarnessJob
|
|
31
|
+
│ │
|
|
32
|
+
▼ ▼
|
|
33
|
+
HarnessExecutor isolated ALK sandbox
|
|
34
|
+
│ + HarnessExecutor
|
|
35
|
+
└──────────── same pipeline ─────────────┘
|
|
36
|
+
│
|
|
37
|
+
▼
|
|
38
|
+
understand → environment → bundle → data → scenarios → connect → simulate → grade
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
`HarnessJob` is the immutable input boundary. Local jobs use a local repository path. Hosted
|
|
42
|
+
jobs use a GitHub installation/repository reference, archive, image, or remote endpoint and can
|
|
43
|
+
never contain a local path. Agent credentials are `SecretRef` values; resolved secrets are
|
|
44
|
+
rejected from serialized jobs.
|
|
45
|
+
|
|
46
|
+
## Environment boundary
|
|
47
|
+
|
|
48
|
+
The environment is infrastructure owned by the harness, not a connection to customer
|
|
49
|
+
production. ALK may adopt schemas, migrations, fixtures and mock services from the submitted
|
|
50
|
+
repository. It then fills missing test data and dependencies itself.
|
|
51
|
+
|
|
52
|
+
Every successful build is sealed as an `EnvironmentBundle`:
|
|
53
|
+
|
|
54
|
+
- versioned schema;
|
|
55
|
+
- SHA-256 content address;
|
|
56
|
+
- exact source/generator provenance;
|
|
57
|
+
- runtime document and service list;
|
|
58
|
+
- named capabilities and readiness probes;
|
|
59
|
+
- per-file hashes and sizes;
|
|
60
|
+
- no symlinks or resolved secrets;
|
|
61
|
+
- immutable verification before execution.
|
|
62
|
+
|
|
63
|
+
This removes repository-path assumptions from the runtime. Local Compose and a future hosted
|
|
64
|
+
Kubernetes/Firecracker provider implement the same `RuntimeProvider` interface and consume the
|
|
65
|
+
same manifest.
|
|
66
|
+
|
|
67
|
+
## Provisioning policy
|
|
68
|
+
|
|
69
|
+
The current local provider follows this order:
|
|
70
|
+
|
|
71
|
+
1. Detect the submitted Compose definition and declared default infrastructure services.
|
|
72
|
+
2. Give the run a unique Compose project and free host ports.
|
|
73
|
+
3. Exclude opt-in agent/worker services from infrastructure startup.
|
|
74
|
+
4. Build and wait for declared health checks once.
|
|
75
|
+
5. Derive only the endpoint configuration the agent already reads.
|
|
76
|
+
6. Reuse a healthy build only when its complete source fingerprint matches.
|
|
77
|
+
7. If no Compose file exists but a Dockerfile does, generate only supported infrastructure
|
|
78
|
+
declared by the contract (currently Postgres, ClickHouse and Redis) and compose it around the
|
|
79
|
+
submitted runtime. Never generate agent tools or proprietary service behavior.
|
|
80
|
+
8. Reset between scenarios from a verified snapshot or isolated lifecycle reset.
|
|
81
|
+
9. Remove the exact project and its test volumes during cleanup.
|
|
82
|
+
|
|
83
|
+
Unknown database engines do not silently fall back to SQLite or Postgres. A store adapter can be
|
|
84
|
+
generated against the engine's native driver, but it must pass generic freeze/restore/counter
|
|
85
|
+
drift and mutation gates before scenarios may use it.
|
|
86
|
+
|
|
87
|
+
## Evidence and grading invariants
|
|
88
|
+
|
|
89
|
+
- Setup calls are never credited to the agent.
|
|
90
|
+
- A missing agent call cannot satisfy a call-dependent check.
|
|
91
|
+
- Checks must fail against an empty or deliberately damaged world.
|
|
92
|
+
- Dependent checks cannot pass when their prerequisite action never happened.
|
|
93
|
+
- Tool refusal, agent failure, simulator failure, connectivity failure, environment failure,
|
|
94
|
+
infrastructure failure and grading failure remain distinct.
|
|
95
|
+
- Transcripts, semantic calls, resulting state, state diffs and recordings are retained according
|
|
96
|
+
to artifact policy.
|
|
97
|
+
- Agent behavior failures are valid RL results; harness/infrastructure failures are not scored as
|
|
98
|
+
agent failures.
|
|
99
|
+
|
|
100
|
+
## Repository-backed chat execution
|
|
101
|
+
|
|
102
|
+
Chat and voice share the same autonomous lifecycle. Voice has a standard realtime rendezvous;
|
|
103
|
+
chat runtimes instead declare the conversational ingress they already implement in the grounded
|
|
104
|
+
agent contract. Today a repository runtime can expose either:
|
|
105
|
+
|
|
106
|
+
- HTTP using ALK's turn envelope; or
|
|
107
|
+
- HTTP using an OpenAI Chat Completions-compatible envelope; or
|
|
108
|
+
- a JSON turn exchange over WebSocket.
|
|
109
|
+
|
|
110
|
+
For every scenario ALK binds the restored world, starts the submitted Compose/Dockerfile/generated
|
|
111
|
+
runtime with the environment's private endpoint overrides, exposes the declared container port on
|
|
112
|
+
an ephemeral loopback port locally (or only the private project network in a hosted runner), waits
|
|
113
|
+
for readiness, drives the conversation, records tool effects and removes the runtime. A default
|
|
114
|
+
Compose API is identified by its declared ingress port and excluded from infrastructure startup;
|
|
115
|
+
ambiguous services fail admission rather than being guessed.
|
|
116
|
+
|
|
117
|
+
If the submitted agent returns model-facing tool calls, ALK executes them against the same world
|
|
118
|
+
and continues the turn with tool results. If the agent executes tools itself through the injected
|
|
119
|
+
environment endpoints, the world records those calls at the service boundary. Either way, setup
|
|
120
|
+
activity remains separate from agent evidence.
|
|
121
|
+
|
|
122
|
+
An agent with no external HTTP/WebSocket ingress is not silently reconstructed. Callable/CLI-only
|
|
123
|
+
repository runtimes need a sandbox-side process adapter in a later extension; until then admission
|
|
124
|
+
reports the missing interface explicitly. This preserves the invariant that ALK runs submitted
|
|
125
|
+
agent behavior rather than inventing it.
|
|
126
|
+
|
|
127
|
+
## Data and scenario quality
|
|
128
|
+
|
|
129
|
+
Scenario validation rejects predictable/demo fixtures such as `123456`, recycled identities,
|
|
130
|
+
and reused payment/booking placeholders. A suite must vary identities, communication styles,
|
|
131
|
+
locations, account/payment states, instructions and expected paths. Submitted seed data is
|
|
132
|
+
preserved where useful and expanded with synthetic records when it is too sparse to exercise the
|
|
133
|
+
contract.
|
|
134
|
+
|
|
135
|
+
The simulator is constrained by literal scenario facts, tracks facts already stated, answers the
|
|
136
|
+
agent's current question, detects rephrased loops, and only retries infrastructure failures.
|
|
137
|
+
Deterministic agent weaknesses remain deterministic failures.
|
|
138
|
+
|
|
139
|
+
## Delivery and recovery
|
|
140
|
+
|
|
141
|
+
All progress uses ALK's canonical, versioned event envelope. `EventOutbox` writes events and
|
|
142
|
+
fsyncs them before attempting upload. Platform delivery is batchable and idempotent by event ID;
|
|
143
|
+
partial acknowledgements leave the remainder pending. A local run therefore completes offline
|
|
144
|
+
and can sync later. Hosted execution uses the same protocol.
|
|
145
|
+
|
|
146
|
+
The existing Future AGI result sink remains responsible for platform run rows, transcripts,
|
|
147
|
+
evaluations and recording upload. Platform views render stored data; they do not reconstruct or
|
|
148
|
+
run harness stages.
|
|
149
|
+
|
|
150
|
+
## Scaling and isolation
|
|
151
|
+
|
|
152
|
+
One job maps to one ephemeral hosted sandbox and one resource envelope. The scheduler may place
|
|
153
|
+
those sandboxes on Kubernetes pods or micro-VMs, but that decision is outside ALK. Required
|
|
154
|
+
production controls are:
|
|
155
|
+
|
|
156
|
+
- dedicated execution cluster/account, never ordinary platform workers;
|
|
157
|
+
- per-job filesystem, network namespace and service identity;
|
|
158
|
+
- deny-by-default egress with explicit provider/GitHub/platform destinations;
|
|
159
|
+
- CPU, memory, disk, duration and concurrency quotas from `RuntimeRequirements`;
|
|
160
|
+
- short-lived repository and provider credentials;
|
|
161
|
+
- no privileged containers or host Docker socket inside untrusted sandboxes;
|
|
162
|
+
- artifact size/retention enforcement;
|
|
163
|
+
- cancellation, orphan reconciliation and guaranteed cleanup;
|
|
164
|
+
- cache only content-addressed dependency/image layers, never mutable customer workspaces.
|
|
165
|
+
|
|
166
|
+
## Extension points
|
|
167
|
+
|
|
168
|
+
- `SourceAcquirer`: GitHub, archive, image or other source materialization.
|
|
169
|
+
- `RuntimeProvider`: local Compose today; isolated hosted provider next.
|
|
170
|
+
- ALK endpoint adapters: callable/local, HTTP, WebSocket, LiveKit, Vapi and Retell today; MCP and
|
|
171
|
+
process/container connectors are the next adapters and must use the same registry.
|
|
172
|
+
- Store registry: Postgres, SQLite and in-process today; generated native adapters for new
|
|
173
|
+
engines after conformance proofs.
|
|
174
|
+
- `EventTransport` and result sinks: local filesystem, Future AGI platform or customer-owned
|
|
175
|
+
telemetry.
|
|
176
|
+
|
|
177
|
+
Adding an environment engine, agent connector, source type or scheduler should be one adapter;
|
|
178
|
+
it must not add a branch to scenario generation or grading.
|
|
179
|
+
|
|
180
|
+
## Agent/tool ownership boundary
|
|
181
|
+
|
|
182
|
+
ALK provisions the environment around submitted agent code; it never supplies missing agent
|
|
183
|
+
behavior. The submitted repository remains authoritative for prompts, tool schemas, tool
|
|
184
|
+
implementations, orchestration and business rules. Environment adaptation is limited to:
|
|
185
|
+
|
|
186
|
+
- starting declared infrastructure such as databases, queues, object stores and media services;
|
|
187
|
+
- injecting non-secret endpoints through configuration seams the submitted code already reads;
|
|
188
|
+
- resolving referenced credentials at runtime;
|
|
189
|
+
- seeding and resetting test-owned dependency state; and
|
|
190
|
+
- capturing calls, tool evidence and generated artifacts without changing their meaning.
|
|
191
|
+
|
|
192
|
+
A customer-specific API or missing tool implementation is not infrastructure. If its
|
|
193
|
+
implementation is absent, admission fails with an unsupported/missing dependency result. The
|
|
194
|
+
harness must not generate a substitute, proxy invented behavior, or grade against its own
|
|
195
|
+
replacement. A code-execution service follows the same rule: ALK may provide an isolated runtime
|
|
196
|
+
when the agent already declares that dependency, but the customer's tool decides what code to
|
|
197
|
+
execute and how its outputs are used.
|
|
198
|
+
|
|
199
|
+
## Admission, credentials and source trust
|
|
200
|
+
|
|
201
|
+
Repository admission is a read-only static preflight. It does not import or execute submitted
|
|
202
|
+
code. The scanner ignores dependencies, tests, generated output, symlinks and oversized files;
|
|
203
|
+
recognizes Python, JavaScript, env-template and Compose declarations; and emits only names,
|
|
204
|
+
purposes and statuses. It models provider alternatives explicitly, so one Gemini API key or one
|
|
205
|
+
complete Vertex credential route satisfies model authentication without asking for every option.
|
|
206
|
+
|
|
207
|
+
Public GitHub source is cloned anonymously. Private source carries a GitHub App installation
|
|
208
|
+
reference; the sandbox's credential broker resolves the short-lived token only for clone and
|
|
209
|
+
injects it through Git configuration environment variables, never process arguments or the job.
|
|
210
|
+
The resolved commit is verified when a commit SHA is supplied. Source fingerprinting hashes
|
|
211
|
+
symlink metadata without following links outside the repository, and the supervisor rejects a
|
|
212
|
+
source tree that changes during execution.
|
|
213
|
+
|
|
214
|
+
Non-secret connector configuration and secret references have separate contracts. Inline secret-
|
|
215
|
+
like configuration keys are rejected by the platform. The worker builds a fresh allowlisted
|
|
216
|
+
environment, resolves only the job's `SecretRef` entries, and does not inherit the supervisor's
|
|
217
|
+
model, cloud, repository or customer credentials.
|
|
218
|
+
|
|
219
|
+
## Retry and ingestion invariants
|
|
220
|
+
|
|
221
|
+
The supervisor retries only structured failures marked retryable in the infrastructure,
|
|
222
|
+
connectivity or platform-sync domains. Each failed worker attempt is archived separately before
|
|
223
|
+
a clean attempt starts. Agent behavior and grading outcomes are never retried to manufacture a
|
|
224
|
+
pass. GitHub clone uses the same bounded exponential-backoff policy.
|
|
225
|
+
|
|
226
|
+
Before terminal success, the artifact directory is sealed with a content-addressed manifest. The
|
|
227
|
+
seal requires a terminal result and non-empty transcript per scenario, validates referenced
|
|
228
|
+
recordings, rejects secret material and unsafe links, enforces the byte budget, and records every
|
|
229
|
+
retained file's SHA-256, media type and size. Platform result payloads and recording uploads carry
|
|
230
|
+
digests. Ingestion locks the call row, accepts identical retries idempotently, rejects conflicting
|
|
231
|
+
evidence, and stores combined, stereo, customer and assistant recordings as distinct artifacts.
|
fi/alk/harness/DESIGN.md
ADDED
|
@@ -0,0 +1,246 @@
|
|
|
1
|
+
# The harness: what it builds and how it proves it
|
|
2
|
+
|
|
3
|
+
The reference for the rebuild. Written after the corrections in `_scenario-generation/context/`
|
|
4
|
+
7.2, 8.1, 9.1, 10.1 and 12, and it supersedes anything in the code that contradicts it.
|
|
5
|
+
|
|
6
|
+
---
|
|
7
|
+
|
|
8
|
+
## The model, in one paragraph
|
|
9
|
+
|
|
10
|
+
The **environment step** builds everything that is common to every test of one agent: the world
|
|
11
|
+
its tools act on, the prompt that drives a simulated user if it has one, and the catalogue of
|
|
12
|
+
sub-goals it can be checked against. Every **scenario** is then only a change on that base: what
|
|
13
|
+
it alters after reset, the instruction substituted into the simulator's prompt, and which
|
|
14
|
+
sub-goals must hold. Nothing about a scenario is a template with slots; the harness writes each
|
|
15
|
+
one, and proves it works before keeping it.
|
|
16
|
+
|
|
17
|
+
```
|
|
18
|
+
environment step ─────────────────────────────► base: world + simulator prompt + sub-goals
|
|
19
|
+
│
|
|
20
|
+
scenario 1 ──► reset → setup → run → check ───────────────┤
|
|
21
|
+
scenario 2 ──► reset → setup → run → check ───────────────┤
|
|
22
|
+
scenario N ──► reset → setup → run → check ───────────────┘
|
|
23
|
+
```
|
|
24
|
+
|
|
25
|
+
---
|
|
26
|
+
|
|
27
|
+
## 1. The environment step
|
|
28
|
+
|
|
29
|
+
> *"You have to first understand what the f\*\*\* this agent is, from that you will create
|
|
30
|
+
> databases, you'll create a snapshot of the databases first."* — Nikhil, 12
|
|
31
|
+
|
|
32
|
+
It produces four things. All of them are written by the harness. None are hardcoded here.
|
|
33
|
+
|
|
34
|
+
### 1.1 The world
|
|
35
|
+
|
|
36
|
+
Whatever **this** agent needs, and nothing more. For the drive-thru agent that is a database.
|
|
37
|
+
For a browser agent it is a site. For something else it is a filesystem, a queue, a service — the
|
|
38
|
+
harness decides from the contract what has to exist.
|
|
39
|
+
|
|
40
|
+
It **subclasses ALK's `EnvironmentAdapter`**, so the runners that already exist can drive it:
|
|
41
|
+
`reset` publishes the tools and the starting state, `handle_tool_call` executes one call, and the
|
|
42
|
+
state afterwards is what gets graded. Nothing the harness writes should re-implement a runner.
|
|
43
|
+
|
|
44
|
+
It is **frozen once** as a snapshot. Every scenario restores from that snapshot, so a run is
|
|
45
|
+
repeatable and no scenario can inherit another's leftovers.
|
|
46
|
+
|
|
47
|
+
### 1.2 The simulator prompt — only where the agent is conversational
|
|
48
|
+
|
|
49
|
+
> *"For voice and chat there is a simulator, and the input is an instruction to that simulator
|
|
50
|
+
> rather than an input to the agent under test. Where there is no actor, variability comes from
|
|
51
|
+
> how the environment is designed."* — 10.1 §4
|
|
52
|
+
|
|
53
|
+
The harness writes one prompt for the simulated user of **this** agent, with variables left open.
|
|
54
|
+
Each scenario supplies the values. The prompt is an artifact of the environment step because it
|
|
55
|
+
is the same for every scenario; only the substituted instruction differs.
|
|
56
|
+
|
|
57
|
+
There is no persona field and no persona library.
|
|
58
|
+
|
|
59
|
+
> *"Drop the gimmicky persona characters. Variability comes from real conditions instead: a new
|
|
60
|
+
> versus an existing user, whether a payment method is on file, addresses."* — 8.1
|
|
61
|
+
|
|
62
|
+
For a browser or coding agent there is no simulator at all; the instruction goes to the agent
|
|
63
|
+
directly.
|
|
64
|
+
|
|
65
|
+
### 1.3 The sub-goal catalogue
|
|
66
|
+
|
|
67
|
+
> *"Defining the sub-goals is our call. The important property is that they are common across
|
|
68
|
+
> scenarios so the results roll up: if a payment step appears in 50 scenarios, the analytics
|
|
69
|
+
> should show where payment fails and how often."* — 10.1 §8
|
|
70
|
+
|
|
71
|
+
Defined **once**, here, as a named list. Scenarios reference them; they do not invent their own
|
|
72
|
+
wording. That is what makes `order-confirmation fails in 7 of 12 scenarios` a sentence anyone can
|
|
73
|
+
say. Each entry carries its own check (see §3).
|
|
74
|
+
|
|
75
|
+
### 1.4 The gate
|
|
76
|
+
|
|
77
|
+
The environment is not accepted because it looks right. It is exercised: every tool called with a
|
|
78
|
+
valid call, a nonexistent id, and a missing argument; sequences where state has to carry across
|
|
79
|
+
calls. **A refusal is the environment working; a crash is a defect.** It cannot be saved dirty
|
|
80
|
+
(rows left over from building) or unverified.
|
|
81
|
+
|
|
82
|
+
---
|
|
83
|
+
|
|
84
|
+
## 2. A scenario is a change on that base, and it owns a folder
|
|
85
|
+
|
|
86
|
+
```
|
|
87
|
+
name identifier, and the name of its folder
|
|
88
|
+
use_case which branch of the agent's real use cases this belongs to
|
|
89
|
+
setup.py def setup(world), what changes after reset. Code, because what a
|
|
90
|
+
scenario changes is not necessarily the database alone
|
|
91
|
+
ready.py def ready(world), whether the world holds what this scenario presumes
|
|
92
|
+
instruction the task. For a conversational agent this is substituted into the
|
|
93
|
+
simulator prompt; for a browser or coding agent it goes to the agent
|
|
94
|
+
solution the reference trajectory: what a correct agent would do
|
|
95
|
+
checks/*.py one file per deterministic sub-goal, each runnable on its own
|
|
96
|
+
```
|
|
97
|
+
|
|
98
|
+
The file is the artifact. Code lives in files rather than as strings inside JSON, and every check
|
|
99
|
+
carries a `__main__` block so a person can run it by hand against what a run left behind and get
|
|
100
|
+
the same answer the harness got.
|
|
101
|
+
|
|
102
|
+
Gone from the old shape: `persona`, `opening`, `goal`, and free-text `must` / `must_not` as the
|
|
103
|
+
primary grading. Scenarios are organised **use case → branch**, not by adversarial flavour.
|
|
104
|
+
|
|
105
|
+
> *"A login flow is not one row with happy/edge inside it; it is many rows: login-with-Google,
|
|
106
|
+
> login-with-Microsoft, forgot-password, sign-up-with-email."* — Nikhil, 7.2
|
|
107
|
+
|
|
108
|
+
---
|
|
109
|
+
|
|
110
|
+
## 3. Checks: deterministic by default, judge as the fallback
|
|
111
|
+
|
|
112
|
+
> *"When you have `==` or a python script, then I'll call that deterministic."*
|
|
113
|
+
> *"Most likely we can make things deterministic."*
|
|
114
|
+
> *"Deterministic, if possible. And LLM also, obviously."* — Nikhil, 10
|
|
115
|
+
|
|
116
|
+
| | |
|
|
117
|
+
|---|---|
|
|
118
|
+
| **Deterministic** — an assert, an equality, **a python script** | The default |
|
|
119
|
+
| **Non-deterministic** — an LLM judging whether a sub-goal was met | Only where nothing observable settles it |
|
|
120
|
+
|
|
121
|
+
The trap, in his words: *"you are judging by LLM [so it is non-deterministic], but if you want an
|
|
122
|
+
exact output to be 50, then that is deterministic."* An exact fact checked by a judge is **still
|
|
123
|
+
non-deterministic**. What matters is who decides, not how precise the fact is.
|
|
124
|
+
|
|
125
|
+
A check is code the harness writes, and it has two observable things to work from:
|
|
126
|
+
|
|
127
|
+
1. **the world afterwards** — rows, files, whatever this environment is
|
|
128
|
+
2. **the recorded tool calls** — that the call happened, *and with the right arguments*
|
|
129
|
+
|
|
130
|
+
That second one answers the question left open in 7.2: a booking made for 10 PM when 11 PM was
|
|
131
|
+
asked for is a failure, and it is deterministic to detect.
|
|
132
|
+
|
|
133
|
+
The judge is left only with what leaves no trace: whether a refusal was explained, whether a price
|
|
134
|
+
was invented, tone.
|
|
135
|
+
|
|
136
|
+
> Warning from the previous run: *"Judge checkpoints, about a third of all checkpoints, are
|
|
137
|
+
> returned as skipped and not graded."* Leaning on the judge does not merely weaken a result — it
|
|
138
|
+
> silently produces holes.
|
|
139
|
+
|
|
140
|
+
---
|
|
141
|
+
|
|
142
|
+
## 4. Three gates on every scenario, before it is kept
|
|
143
|
+
|
|
144
|
+
Terminal-bench's oracle run, which is the reason its tasks are known to be solvable.
|
|
145
|
+
|
|
146
|
+
### Gate 1. Ready
|
|
147
|
+
|
|
148
|
+
```
|
|
149
|
+
reset → apply setup → run ready ⇒ must HOLD
|
|
150
|
+
```
|
|
151
|
+
|
|
152
|
+
The world must hold what the scenario presumes. A scenario about the last five items is only a
|
|
153
|
+
test of the agent if there really are five; otherwise the agent fails for a precondition we got
|
|
154
|
+
wrong, and the report reads as a finding about the agent. A missing precondition is ours, and this
|
|
155
|
+
is where it is caught.
|
|
156
|
+
|
|
157
|
+
### Gate 2. Solvable
|
|
158
|
+
|
|
159
|
+
```
|
|
160
|
+
reset → apply setup → run the solution → run the checks ⇒ must PASS
|
|
161
|
+
```
|
|
162
|
+
|
|
163
|
+
If the checks fail with the reference solution, either the scenario is impossible or the check is
|
|
164
|
+
wrong. Both have already happened here: a scenario asserted a value the agent was never permitted
|
|
165
|
+
to send, and another demanded confirmation of an item that could not be ordered. This catches
|
|
166
|
+
them at write time, with no model involved.
|
|
167
|
+
|
|
168
|
+
### Gate 3. Not vacuous
|
|
169
|
+
|
|
170
|
+
```
|
|
171
|
+
reset → apply setup → run NOTHING → run the checks ⇒ must FAIL
|
|
172
|
+
```
|
|
173
|
+
|
|
174
|
+
A check that passes without the agent doing anything grades nothing while reporting a result.
|
|
175
|
+
This is the failure that makes a suite quietly green.
|
|
176
|
+
|
|
177
|
+
Vacuity is judged on *all* checks passing, because one check surviving an empty run ("no
|
|
178
|
+
unavailable item was ordered") is legitimate. A single check that survives is still named, because
|
|
179
|
+
sub-goals are shared: a check that cannot fail without calls would roll up as a pass for an agent
|
|
180
|
+
that did nothing at all.
|
|
181
|
+
|
|
182
|
+
Neither gate asks a model anything. The environment decides.
|
|
183
|
+
|
|
184
|
+
**Three things fall out of the solution for free:** it is the expected trajectory; comparing the
|
|
185
|
+
agent's trajectory against it gives efficiency (Nikhil's point about the agent that succeeds on
|
|
186
|
+
the 21st call after 20 failures); and a scenario that cannot be run is caught before a call is
|
|
187
|
+
ever placed.
|
|
188
|
+
|
|
189
|
+
---
|
|
190
|
+
|
|
191
|
+
## 5. Running it
|
|
192
|
+
|
|
193
|
+
> *"Use an existing harness. Just for Claude agents. Use any existing harness that is there."*
|
|
194
|
+
> — Nikhil, 12
|
|
195
|
+
|
|
196
|
+
The simulation runs through **ALK's own path**, not a loop written here. The world is passed in
|
|
197
|
+
as the environment; the agent under test is the real agent, in its real runtime. For the voice
|
|
198
|
+
case that means the Vapi assistant we already have, over LiveKit, with the tool webhook answered
|
|
199
|
+
by **our world** rather than by canned mocks.
|
|
200
|
+
|
|
201
|
+
That last part is the whole point of the environment. The previous run's known issues were:
|
|
202
|
+
|
|
203
|
+
- *"Mocked tools always succeed, including removing an item that was never added."*
|
|
204
|
+
- *"Mock responses do not vary by argument, so read-after-write flows are wrong."*
|
|
205
|
+
- *"World state does not change unless a scenario sets `state_updates`, which is often empty."*
|
|
206
|
+
|
|
207
|
+
A world that really holds rows and can really refuse removes all three.
|
|
208
|
+
|
|
209
|
+
---
|
|
210
|
+
|
|
211
|
+
## 6. What changes per kind of agent, and what does not
|
|
212
|
+
|
|
213
|
+
| | Voice / chat | Browser | Coding |
|
|
214
|
+
|---|---|---|---|
|
|
215
|
+
| World | database, KB | a site | a filesystem, a repo |
|
|
216
|
+
| Simulated user | yes — prompt written by the harness | none | none |
|
|
217
|
+
| Instruction goes to | the simulator | the agent | the agent |
|
|
218
|
+
| Solution | tool calls | actions | commands |
|
|
219
|
+
| Check | code over world + calls | code over the page + actions | code over the tree |
|
|
220
|
+
|
|
221
|
+
**What never changes:** the environment is built once and frozen; a scenario is a change on it; a
|
|
222
|
+
solution proves it is solvable; a scenario whose world is not ready is rejected before it can be
|
|
223
|
+
blamed on the agent; a check that cannot fail is rejected; deterministic first.
|
|
224
|
+
|
|
225
|
+
---
|
|
226
|
+
|
|
227
|
+
## 7. Order of work
|
|
228
|
+
|
|
229
|
+
1. **Environment step** — the world, the simulator prompt, the sub-goal catalogue, all written by
|
|
230
|
+
the harness rather than by a fixed schema here.
|
|
231
|
+
2. **Scenario shape**: a folder per scenario: setup / ready / instruction / solution / sub-goal
|
|
232
|
+
references / a file per check.
|
|
233
|
+
3. **The three gates**: ready, solvable, and not vacuous.
|
|
234
|
+
4. **Run through ALK** — the world serving the tool calls of the real agent.
|
|
235
|
+
|
|
236
|
+
---
|
|
237
|
+
|
|
238
|
+
## 8. Instructions, not code
|
|
239
|
+
|
|
240
|
+
> *"This is a flow, this is not a harness. You will give your harness the instructions that you
|
|
241
|
+
> are supposed to do all this and then the harness will do all that. It's not a code that your
|
|
242
|
+
> harness follows."* — Nikhil, 12
|
|
243
|
+
|
|
244
|
+
Every stage's method lives in a `SKILL.md`, editable without touching code. What stays in code is
|
|
245
|
+
only what must be exact: executing a call, restoring a snapshot, running a check, and refusing
|
|
246
|
+
something that does not hold up. **The harness decides what to do. Code decides what is true.**
|