agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
"""Stage four: run the scenarios against the real agent, and say what came back.
|
|
2
|
+
|
|
3
|
+
The last stage that was a command rather than a conversation. Nothing about it needed to be:
|
|
4
|
+
wiring the world to the assistant and running the checks is already code, and the part worth
|
|
5
|
+
having judgement on is which scenario to run and what a failure actually means.
|
|
6
|
+
|
|
7
|
+
That second part is why this is a stage at all. A failing check has four possible causes and only
|
|
8
|
+
one of them is a finding about the agent — the others are a wrong check, a wrong contract, or a
|
|
9
|
+
simulated caller that never asked for the thing. Deciding which is reading, not arithmetic.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
from typing import Any, Callable
|
|
16
|
+
|
|
17
|
+
from ..backends import SessionSpec
|
|
18
|
+
from ..config import artifact_dir, chosen_model, load_skill
|
|
19
|
+
from ..contract import AgentContract
|
|
20
|
+
from ..scenario_tools import load_scenarios
|
|
21
|
+
from ..session import Stage
|
|
22
|
+
from .tools import (
|
|
23
|
+
RUN_SERVER,
|
|
24
|
+
load_results,
|
|
25
|
+
missing_prerequisites,
|
|
26
|
+
run_tools,
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
SKILL = "run-scenarios"
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def open_stage(
|
|
33
|
+
contract: AgentContract,
|
|
34
|
+
*,
|
|
35
|
+
out: Path | None = None,
|
|
36
|
+
ask: Callable[..., Any] | None = None,
|
|
37
|
+
max_turns: int = 40,
|
|
38
|
+
) -> tuple[Stage, Path]:
|
|
39
|
+
"""A live run-the-scenarios stage, and where it will write its results."""
|
|
40
|
+
destination = out or artifact_dir(contract.agent)
|
|
41
|
+
server = run_tools(destination, destination, contract=contract)
|
|
42
|
+
spec = SessionSpec(
|
|
43
|
+
system_prompt=(f"{load_skill(SKILL)}\n\n## This agent\n\n{contract.brief()}"),
|
|
44
|
+
servers={RUN_SERVER: server},
|
|
45
|
+
builtins=("AskUserQuestion",),
|
|
46
|
+
cwd=str(destination.parent if destination.parent.exists() else Path.cwd()),
|
|
47
|
+
max_turns=max_turns,
|
|
48
|
+
model=chosen_model(),
|
|
49
|
+
ask=ask,
|
|
50
|
+
)
|
|
51
|
+
return Stage(spec, name=SKILL), destination
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def opening(contract: AgentContract, destination: Path) -> str:
|
|
55
|
+
"""What to tell the stage when it opens.
|
|
56
|
+
|
|
57
|
+
Deliberately does not tell it to run everything. Each call costs money and takes minutes, and
|
|
58
|
+
a stage that opens by spending the whole suite gives nobody a chance to say which one they
|
|
59
|
+
cared about.
|
|
60
|
+
"""
|
|
61
|
+
written = load_scenarios(destination)
|
|
62
|
+
already = load_results(destination)
|
|
63
|
+
blocked = (
|
|
64
|
+
missing_prerequisites(destination, contract)
|
|
65
|
+
if contract.modality == "voice"
|
|
66
|
+
else []
|
|
67
|
+
)
|
|
68
|
+
if blocked:
|
|
69
|
+
return (
|
|
70
|
+
f"There are {len(written)} scenarios for {contract.agent!r}, but a live call cannot "
|
|
71
|
+
"be placed yet:\n - "
|
|
72
|
+
+ "\n - ".join(blocked)
|
|
73
|
+
+ "\n\nSay this plainly and stop."
|
|
74
|
+
)
|
|
75
|
+
if already:
|
|
76
|
+
passed = sum(1 for record in already if record["passed"])
|
|
77
|
+
return (
|
|
78
|
+
f"{len(already)} of {len(written)} scenarios for {contract.agent!r} have been run, "
|
|
79
|
+
f"{passed} passing. Say where things stand with read_results, then ask which to run."
|
|
80
|
+
)
|
|
81
|
+
return (
|
|
82
|
+
f"{len(written)} scenarios are ready for {contract.agent!r} and none has been run.\n\n"
|
|
83
|
+
"Run preflight, then list_scenarios, then say which ones you would run first and why. "
|
|
84
|
+
"Do not start running them until you are asked to — each call takes minutes and costs "
|
|
85
|
+
"real money."
|
|
86
|
+
)
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def load(destination: Path) -> list[dict[str, Any]]:
|
|
90
|
+
"""What has been run for this agent, if anything has."""
|
|
91
|
+
return load_results(Path(destination))
|
|
@@ -0,0 +1,508 @@
|
|
|
1
|
+
"""What is being tested, and how the harness talks to it.
|
|
2
|
+
|
|
3
|
+
The rest of the run does not care what the agent under test is. It says something and gets a
|
|
4
|
+
reply back, and whatever tool calls happened in between landed in the world. That is the entire
|
|
5
|
+
interface, and keeping it that narrow is what lets the same scenarios, the same world and the
|
|
6
|
+
same grading run against an agent hosted anywhere.
|
|
7
|
+
|
|
8
|
+
Two things are supplied per target: how to say something to it, and how its tool calls reach the
|
|
9
|
+
world. ``LocalAgent`` is only for contract-only specs with no submitted implementation.
|
|
10
|
+
``RepositoryChatTarget`` starts the submitted runtime and reaches its existing HTTP/WebSocket
|
|
11
|
+
interface. A hosted target uses the same narrow protocol with the transport swapped: its tool
|
|
12
|
+
calls arrive over a webhook or in a turn response, and the same world answers them. The world,
|
|
13
|
+
scenarios and grading do not change.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import asyncio
|
|
19
|
+
import json
|
|
20
|
+
from pathlib import Path
|
|
21
|
+
import socket
|
|
22
|
+
import time
|
|
23
|
+
from typing import Any, Callable, Protocol, runtime_checkable
|
|
24
|
+
from urllib.parse import urljoin
|
|
25
|
+
|
|
26
|
+
from ..backends import SessionSpec, resolve as resolve_backend, tool, tool_server
|
|
27
|
+
|
|
28
|
+
from ..config import chosen_model
|
|
29
|
+
from ..contract import AgentContract
|
|
30
|
+
from ..session import Stage
|
|
31
|
+
from ..world.runtime import GeneratedWorld
|
|
32
|
+
|
|
33
|
+
AGENT_SERVER = "agent"
|
|
34
|
+
|
|
35
|
+
_TYPES: dict[str, type] = {
|
|
36
|
+
"str": str,
|
|
37
|
+
"string": str,
|
|
38
|
+
"int": int,
|
|
39
|
+
"integer": int,
|
|
40
|
+
"float": float,
|
|
41
|
+
"number": float,
|
|
42
|
+
"bool": bool,
|
|
43
|
+
"boolean": bool,
|
|
44
|
+
"list": list,
|
|
45
|
+
"dict": dict,
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _python_type(declared: str) -> type:
|
|
50
|
+
"""The type a tool's argument is declared with, as something a schema can carry."""
|
|
51
|
+
lowered = (declared or "").strip().lower()
|
|
52
|
+
if lowered.startswith(("list", "sequence", "array")):
|
|
53
|
+
return list
|
|
54
|
+
if lowered.startswith(("dict", "mapping", "object")):
|
|
55
|
+
return dict
|
|
56
|
+
return _TYPES.get(lowered, str)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def describe(spec: Any, contract: AgentContract) -> str:
|
|
60
|
+
"""What the agent is told a tool takes, including the values it accepts.
|
|
61
|
+
|
|
62
|
+
The values matter more than they look. An agent whose real schema enumerates its menu knows
|
|
63
|
+
that a Big Mac combo is ``big_mac_combo``; the same agent without them guesses, gets refused,
|
|
64
|
+
and reads as broken when what is broken is the harness that withheld them. Anything the
|
|
65
|
+
contract recorded as permitted, the agent under test is told.
|
|
66
|
+
"""
|
|
67
|
+
parts = [spec.description or f"{spec.name} for {contract.agent}"]
|
|
68
|
+
for arg in spec.args:
|
|
69
|
+
values = spec.arg_values.get(arg)
|
|
70
|
+
if isinstance(values, (list, tuple)) and values:
|
|
71
|
+
rendered = ", ".join(str(value) for value in values)
|
|
72
|
+
parts.append(f" {arg} accepts: {rendered}")
|
|
73
|
+
elif arg in spec.arg_types:
|
|
74
|
+
parts.append(f" {arg}: {spec.arg_types[arg]}")
|
|
75
|
+
return "\n".join(parts)
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def agent_tools(contract: AgentContract, world: GeneratedWorld) -> Any:
|
|
79
|
+
"""The agent's own tools, wired to the world so a call really happens.
|
|
80
|
+
|
|
81
|
+
Every call goes through ``world.call``, so a refusal comes back as a refusal the agent can
|
|
82
|
+
read and recover from, rather than as a success it will happily build on.
|
|
83
|
+
"""
|
|
84
|
+
|
|
85
|
+
def bind(spec: Any) -> Any:
|
|
86
|
+
schema = {
|
|
87
|
+
arg: _python_type(spec.arg_types.get(arg, "str")) for arg in spec.args
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
@tool(spec.name, describe(spec, contract), schema)
|
|
91
|
+
async def call_tool(
|
|
92
|
+
args: dict[str, Any], _name: str = spec.name
|
|
93
|
+
) -> dict[str, Any]:
|
|
94
|
+
# Through handle_tool_call, not straight to world.call. That method is the interface
|
|
95
|
+
# ALK's own runners drive an environment by, so going around it would leave the
|
|
96
|
+
# claim that a generated world plugs into them untested — and free to drift.
|
|
97
|
+
done = world.handle_tool_call({"name": _name, "arguments": args})
|
|
98
|
+
if done is None:
|
|
99
|
+
return {
|
|
100
|
+
"content": [{"type": "text", "text": f"no such tool {_name}"}],
|
|
101
|
+
"is_error": True,
|
|
102
|
+
}
|
|
103
|
+
return {
|
|
104
|
+
"content": [{"type": "text", "text": done.content or ""}],
|
|
105
|
+
**({} if done.success else {"is_error": True}),
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
return call_tool
|
|
109
|
+
|
|
110
|
+
return tool_server(
|
|
111
|
+
name=AGENT_SERVER,
|
|
112
|
+
version="0.1.0",
|
|
113
|
+
tools=[bind(spec) for spec in contract.tools],
|
|
114
|
+
)
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def agent_prompt(contract: AgentContract) -> str:
|
|
118
|
+
"""The agent under test, as its contract describes it.
|
|
119
|
+
|
|
120
|
+
Only what the contract records, because anything added here is a difference between the agent
|
|
121
|
+
being graded and the agent that exists.
|
|
122
|
+
"""
|
|
123
|
+
parts = [
|
|
124
|
+
f"You are {contract.agent}: {contract.one_liner}".strip(),
|
|
125
|
+
contract.system_prompt_excerpt.strip(),
|
|
126
|
+
]
|
|
127
|
+
if contract.hard_constraints:
|
|
128
|
+
parts.append(
|
|
129
|
+
"Rules you must follow:\n - " + "\n - ".join(contract.hard_constraints)
|
|
130
|
+
)
|
|
131
|
+
if contract.modality == "voice":
|
|
132
|
+
parts.append(
|
|
133
|
+
"You are speaking out loud. Keep replies to what a person would actually say: "
|
|
134
|
+
"short, no lists, no markdown."
|
|
135
|
+
)
|
|
136
|
+
parts.append(
|
|
137
|
+
"Use your tools to do anything real. Never tell the customer something is done unless a "
|
|
138
|
+
"tool confirmed it, and if a tool refuses, say so plainly and offer what is possible."
|
|
139
|
+
)
|
|
140
|
+
return "\n\n".join(part for part in parts if part)
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
@runtime_checkable
|
|
144
|
+
class Target(Protocol):
|
|
145
|
+
"""An agent under test, reachable by saying something to it."""
|
|
146
|
+
|
|
147
|
+
key: str
|
|
148
|
+
|
|
149
|
+
async def open(self) -> None: ...
|
|
150
|
+
async def say(self, utterance: str) -> str: ...
|
|
151
|
+
async def close(self) -> None: ...
|
|
152
|
+
@property
|
|
153
|
+
def spent_usd(self) -> float: ...
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def _drivable(model: str | None) -> None:
|
|
157
|
+
"""Refuse a model the selected backend cannot actually run, before a suite is graded on it.
|
|
158
|
+
|
|
159
|
+
Handed a model it cannot reach, a backend does not fail: it produces a session that answers
|
|
160
|
+
nothing, which arrives as a scenario with no turns and no calls and every check red. That
|
|
161
|
+
reads exactly like an agent that ignored the person, and the whole suite is wrong in a way
|
|
162
|
+
nobody would think to question.
|
|
163
|
+
"""
|
|
164
|
+
named = (model or "").strip().lower()
|
|
165
|
+
if not named:
|
|
166
|
+
return
|
|
167
|
+
backend = resolve_backend()
|
|
168
|
+
if backend.can_drive(named):
|
|
169
|
+
return
|
|
170
|
+
raise RuntimeError(
|
|
171
|
+
f"this target cannot run {model!r}. The selected harness backend "
|
|
172
|
+
f"({backend.name}) does not drive that model. Pick a model that backend serves, "
|
|
173
|
+
"select the backend that serves it through ALK_HARNESS, or point the spec's target "
|
|
174
|
+
"at one of ALK's own endpoint adapters rather than at this one."
|
|
175
|
+
)
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
class LocalAgent:
|
|
179
|
+
"""The agent run here, from its contract, with its tools bound to the world."""
|
|
180
|
+
|
|
181
|
+
key = "local"
|
|
182
|
+
|
|
183
|
+
def __init__(
|
|
184
|
+
self,
|
|
185
|
+
contract: AgentContract,
|
|
186
|
+
world: GeneratedWorld,
|
|
187
|
+
*,
|
|
188
|
+
model: str | None = None,
|
|
189
|
+
max_turns: int = 12,
|
|
190
|
+
) -> None:
|
|
191
|
+
self.contract = contract
|
|
192
|
+
self.world = world
|
|
193
|
+
if contract.runtime or contract.tool_entrypoints or contract.implementation:
|
|
194
|
+
raise RuntimeError(
|
|
195
|
+
"the local contract target is disabled for repository-backed agents because it "
|
|
196
|
+
"reconstructs the agent from its prompt. Register a target that starts the "
|
|
197
|
+
"agent's shipped runtime and applies the provisioned endpoint overrides."
|
|
198
|
+
)
|
|
199
|
+
_drivable(model)
|
|
200
|
+
# The agent under test gets its own tools and nothing else. A target that can reach a
|
|
201
|
+
# file or a shell is not the agent anybody deployed.
|
|
202
|
+
spec = SessionSpec(
|
|
203
|
+
system_prompt=agent_prompt(contract),
|
|
204
|
+
servers={AGENT_SERVER: agent_tools(contract, world)},
|
|
205
|
+
max_turns=max_turns,
|
|
206
|
+
model=chosen_model(model),
|
|
207
|
+
)
|
|
208
|
+
self._stage = Stage(spec, name="target")
|
|
209
|
+
|
|
210
|
+
async def open(self) -> None:
|
|
211
|
+
await self._stage.__aenter__()
|
|
212
|
+
|
|
213
|
+
async def say(self, utterance: str) -> str:
|
|
214
|
+
turn = await self._stage.say(utterance)
|
|
215
|
+
return turn.text.strip()
|
|
216
|
+
|
|
217
|
+
async def close(self) -> None:
|
|
218
|
+
await self._stage.__aexit__(None, None, None)
|
|
219
|
+
|
|
220
|
+
@property
|
|
221
|
+
def spent_usd(self) -> float:
|
|
222
|
+
return self._stage.spent_usd
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
class RepositoryChatTarget:
|
|
226
|
+
"""The submitted chat runtime, reached over its existing turn ingress.
|
|
227
|
+
|
|
228
|
+
Lifecycle mirrors the repository-backed voice path: bind the scenario's generated world,
|
|
229
|
+
start the isolated submitted runtime with only endpoint substitutions, wait for its real
|
|
230
|
+
ingress, converse, then remove only that runtime. The source prompt and tools are never
|
|
231
|
+
reconstructed in this process.
|
|
232
|
+
"""
|
|
233
|
+
|
|
234
|
+
key = "repository"
|
|
235
|
+
|
|
236
|
+
def __init__(
|
|
237
|
+
self,
|
|
238
|
+
contract: AgentContract,
|
|
239
|
+
world: GeneratedWorld,
|
|
240
|
+
*,
|
|
241
|
+
world_root: str | Path,
|
|
242
|
+
trace_path: str | Path | None = None,
|
|
243
|
+
scenario_name: str = "scenario",
|
|
244
|
+
) -> None:
|
|
245
|
+
runtime = contract.runtime
|
|
246
|
+
interface = runtime.interface if runtime is not None else None
|
|
247
|
+
if interface is None:
|
|
248
|
+
raise RuntimeError(
|
|
249
|
+
"the submitted chat runtime has no recorded conversational interface. "
|
|
250
|
+
"Record its existing HTTP port, path and protocol during understanding; the "
|
|
251
|
+
"harness will not reconstruct the repository agent from its prompt."
|
|
252
|
+
)
|
|
253
|
+
if interface.kind not in {"http", "websocket"}:
|
|
254
|
+
raise RuntimeError(
|
|
255
|
+
f"submitted chat interface {interface.kind!r} is not wired for hosted repository "
|
|
256
|
+
"execution yet; supported now: HTTP and WebSocket"
|
|
257
|
+
)
|
|
258
|
+
if interface.port is None:
|
|
259
|
+
raise RuntimeError("submitted HTTP chat interface has no container port")
|
|
260
|
+
self.contract = contract
|
|
261
|
+
self.world = world
|
|
262
|
+
self.world_root = Path(world_root)
|
|
263
|
+
self.trace_path = Path(trace_path) if trace_path is not None else None
|
|
264
|
+
self.scenario_name = scenario_name
|
|
265
|
+
self.interface = interface
|
|
266
|
+
self._webhook: Any | None = None
|
|
267
|
+
self._wrapper: Any | None = None
|
|
268
|
+
self._messages: list[dict[str, Any]] = []
|
|
269
|
+
self._turn = 0
|
|
270
|
+
|
|
271
|
+
async def open(self) -> None:
|
|
272
|
+
from ..provision import (
|
|
273
|
+
connect_runner_network,
|
|
274
|
+
runtime_endpoint,
|
|
275
|
+
start_runtime,
|
|
276
|
+
)
|
|
277
|
+
from .voice import WorldWebhook
|
|
278
|
+
|
|
279
|
+
webhook = WorldWebhook().start()
|
|
280
|
+
webhook.bind(self.world)
|
|
281
|
+
self._webhook = webhook
|
|
282
|
+
try:
|
|
283
|
+
private_host = await asyncio.to_thread(
|
|
284
|
+
connect_runner_network, self.world_root
|
|
285
|
+
)
|
|
286
|
+
tool_url = (
|
|
287
|
+
f"http://{private_host}:{webhook.port}"
|
|
288
|
+
if private_host
|
|
289
|
+
else f"http://host.docker.internal:{webhook.port}"
|
|
290
|
+
)
|
|
291
|
+
await asyncio.to_thread(
|
|
292
|
+
start_runtime,
|
|
293
|
+
self.world_root,
|
|
294
|
+
overrides={"TOOLS_API_URL": tool_url},
|
|
295
|
+
trace_path=self.trace_path,
|
|
296
|
+
publish_ports=[self.interface.port],
|
|
297
|
+
stable_seconds=0.5,
|
|
298
|
+
)
|
|
299
|
+
# A runtime-only Compose project creates its network at start. This second call is
|
|
300
|
+
# the same idempotent attach used by voice and makes its private container address
|
|
301
|
+
# reachable from a hosted runner container.
|
|
302
|
+
await asyncio.to_thread(connect_runner_network, self.world_root)
|
|
303
|
+
scheme = "ws" if self.interface.kind == "websocket" else "http"
|
|
304
|
+
base = await asyncio.to_thread(
|
|
305
|
+
runtime_endpoint,
|
|
306
|
+
self.world_root,
|
|
307
|
+
self.interface.port,
|
|
308
|
+
scheme=scheme,
|
|
309
|
+
)
|
|
310
|
+
await asyncio.to_thread(self._wait_ready, base)
|
|
311
|
+
if self.interface.kind == "websocket":
|
|
312
|
+
from fi.simulate.agent.wrappers.websocket import WebSocketAgentWrapper
|
|
313
|
+
|
|
314
|
+
wrapper = WebSocketAgentWrapper
|
|
315
|
+
else:
|
|
316
|
+
from fi.simulate.agent.wrappers.http import HTTPAgentWrapper
|
|
317
|
+
|
|
318
|
+
wrapper = HTTPAgentWrapper
|
|
319
|
+
self._wrapper = wrapper(
|
|
320
|
+
endpoint=urljoin(
|
|
321
|
+
base.rstrip("/") + "/", self.interface.path.lstrip("/")
|
|
322
|
+
),
|
|
323
|
+
protocol=self.interface.protocol,
|
|
324
|
+
include_tools=self.interface.include_tools,
|
|
325
|
+
timeout=30.0,
|
|
326
|
+
metadata={
|
|
327
|
+
"target": "submitted_repository_runtime",
|
|
328
|
+
"scenario": self.scenario_name,
|
|
329
|
+
},
|
|
330
|
+
)
|
|
331
|
+
except Exception:
|
|
332
|
+
await self.close()
|
|
333
|
+
raise
|
|
334
|
+
|
|
335
|
+
def _wait_ready(self, base: str) -> None:
|
|
336
|
+
from urllib import error as urllib_error
|
|
337
|
+
from urllib import request as urllib_request
|
|
338
|
+
from urllib.parse import urlsplit
|
|
339
|
+
|
|
340
|
+
deadline = time.monotonic() + 60.0
|
|
341
|
+
health = self.interface.health_path
|
|
342
|
+
last = "not reachable"
|
|
343
|
+
while time.monotonic() < deadline:
|
|
344
|
+
try:
|
|
345
|
+
if health and self.interface.kind == "http":
|
|
346
|
+
url = urljoin(base.rstrip("/") + "/", health.lstrip("/"))
|
|
347
|
+
with urllib_request.urlopen(url, timeout=2) as response:
|
|
348
|
+
if int(getattr(response, "status", 200)) < 500:
|
|
349
|
+
return
|
|
350
|
+
else:
|
|
351
|
+
parsed = urlsplit(base)
|
|
352
|
+
with socket.create_connection(
|
|
353
|
+
(str(parsed.hostname), int(parsed.port or 80)), timeout=2
|
|
354
|
+
):
|
|
355
|
+
return
|
|
356
|
+
except (OSError, urllib_error.URLError) as exc:
|
|
357
|
+
last = f"{type(exc).__name__}: {exc}"
|
|
358
|
+
time.sleep(0.25)
|
|
359
|
+
raise RuntimeError(
|
|
360
|
+
f"submitted chat runtime did not become ready on port {self.interface.port}: {last}"
|
|
361
|
+
)
|
|
362
|
+
|
|
363
|
+
async def say(self, utterance: str) -> str:
|
|
364
|
+
if self._wrapper is None:
|
|
365
|
+
raise RuntimeError("submitted chat runtime is not open")
|
|
366
|
+
from fi.simulate.agent.wrapper import AgentInput
|
|
367
|
+
|
|
368
|
+
self._messages.append({"role": "user", "content": utterance})
|
|
369
|
+
for _step in range(8):
|
|
370
|
+
request = AgentInput(
|
|
371
|
+
thread_id=self.scenario_name,
|
|
372
|
+
execution_id=self.scenario_name,
|
|
373
|
+
turn_index=self._turn,
|
|
374
|
+
scenario_name=self.scenario_name,
|
|
375
|
+
modality="text",
|
|
376
|
+
messages=list(self._messages),
|
|
377
|
+
new_message=dict(self._messages[-1]),
|
|
378
|
+
tools=self._tools() if self.interface.include_tools else [],
|
|
379
|
+
)
|
|
380
|
+
response = await self._wrapper.call(request)
|
|
381
|
+
trace = dict((response.metadata or {}).get("external_agent") or {})
|
|
382
|
+
if trace and not trace.get("success", False):
|
|
383
|
+
raise RuntimeError(
|
|
384
|
+
str(trace.get("error") or "submitted chat endpoint request failed")
|
|
385
|
+
)
|
|
386
|
+
calls = list(response.tool_calls or [])
|
|
387
|
+
if calls:
|
|
388
|
+
self._messages.append(
|
|
389
|
+
{
|
|
390
|
+
"role": "assistant",
|
|
391
|
+
"content": response.content or "",
|
|
392
|
+
"tool_calls": calls,
|
|
393
|
+
}
|
|
394
|
+
)
|
|
395
|
+
for index, call in enumerate(calls, start=1):
|
|
396
|
+
name, arguments, call_id = self._tool_call(call, index)
|
|
397
|
+
result = self.world.handle_tool_call(
|
|
398
|
+
{"id": call_id, "name": name, "arguments": arguments}
|
|
399
|
+
)
|
|
400
|
+
self._messages.append(
|
|
401
|
+
{
|
|
402
|
+
"role": "tool",
|
|
403
|
+
"tool_call_id": call_id,
|
|
404
|
+
"name": name,
|
|
405
|
+
"content": (
|
|
406
|
+
result.content
|
|
407
|
+
if result is not None
|
|
408
|
+
else f"there is no tool called {name}"
|
|
409
|
+
),
|
|
410
|
+
}
|
|
411
|
+
)
|
|
412
|
+
continue
|
|
413
|
+
said = response.content.strip()
|
|
414
|
+
self._messages.append({"role": "assistant", "content": said})
|
|
415
|
+
self._turn += 1
|
|
416
|
+
return said
|
|
417
|
+
raise RuntimeError(
|
|
418
|
+
"submitted chat agent exceeded 8 tool continuations in one turn"
|
|
419
|
+
)
|
|
420
|
+
|
|
421
|
+
@staticmethod
|
|
422
|
+
def _tool_call(call: dict[str, Any], index: int) -> tuple[str, dict[str, Any], str]:
|
|
423
|
+
function = (
|
|
424
|
+
call.get("function") if isinstance(call.get("function"), dict) else {}
|
|
425
|
+
)
|
|
426
|
+
name = str(call.get("name") or function.get("name") or "")
|
|
427
|
+
raw = call.get("arguments", function.get("arguments", {}))
|
|
428
|
+
if isinstance(raw, str):
|
|
429
|
+
try:
|
|
430
|
+
parsed = json.loads(raw)
|
|
431
|
+
except ValueError:
|
|
432
|
+
parsed = {"_raw": raw}
|
|
433
|
+
else:
|
|
434
|
+
parsed = dict(raw or {}) if isinstance(raw, dict) else {}
|
|
435
|
+
return name, parsed, str(call.get("id") or f"call_{index}")
|
|
436
|
+
|
|
437
|
+
def _tools(self) -> list[dict[str, Any]]:
|
|
438
|
+
tools: list[dict[str, Any]] = []
|
|
439
|
+
for spec in self.contract.tools:
|
|
440
|
+
properties = {
|
|
441
|
+
arg: {"type": self._json_type(spec.arg_types.get(arg, "string"))}
|
|
442
|
+
for arg in spec.args
|
|
443
|
+
}
|
|
444
|
+
tools.append(
|
|
445
|
+
{
|
|
446
|
+
"name": spec.name,
|
|
447
|
+
"description": spec.description,
|
|
448
|
+
"parameters": {
|
|
449
|
+
"type": "object",
|
|
450
|
+
"properties": properties,
|
|
451
|
+
"required": list(spec.args),
|
|
452
|
+
},
|
|
453
|
+
}
|
|
454
|
+
)
|
|
455
|
+
return tools
|
|
456
|
+
|
|
457
|
+
@staticmethod
|
|
458
|
+
def _json_type(declared: str) -> str:
|
|
459
|
+
normalized = str(declared or "").lower()
|
|
460
|
+
if any(mark in normalized for mark in ("int", "float", "number")):
|
|
461
|
+
return "number"
|
|
462
|
+
if "bool" in normalized:
|
|
463
|
+
return "boolean"
|
|
464
|
+
if any(mark in normalized for mark in ("list", "array", "sequence")):
|
|
465
|
+
return "array"
|
|
466
|
+
if any(mark in normalized for mark in ("dict", "map", "object")):
|
|
467
|
+
return "object"
|
|
468
|
+
return "string"
|
|
469
|
+
|
|
470
|
+
async def close(self) -> None:
|
|
471
|
+
from ..provision import stop_runtime
|
|
472
|
+
|
|
473
|
+
try:
|
|
474
|
+
await asyncio.to_thread(stop_runtime, self.world_root)
|
|
475
|
+
finally:
|
|
476
|
+
if self._webhook is not None:
|
|
477
|
+
self._webhook.stop()
|
|
478
|
+
self._webhook = None
|
|
479
|
+
self._wrapper = None
|
|
480
|
+
|
|
481
|
+
@property
|
|
482
|
+
def spent_usd(self) -> float:
|
|
483
|
+
# The submitted target owns its model/provider accounting. Provider evidence may add it
|
|
484
|
+
# later; the harness must not fabricate a cost from HTTP traffic.
|
|
485
|
+
return 0.0
|
|
486
|
+
|
|
487
|
+
|
|
488
|
+
_REGISTRY: dict[str, Callable[..., Target]] = {
|
|
489
|
+
LocalAgent.key: LocalAgent,
|
|
490
|
+
RepositoryChatTarget.key: RepositoryChatTarget,
|
|
491
|
+
}
|
|
492
|
+
|
|
493
|
+
|
|
494
|
+
def register_target(key: str, factory: Callable[..., Target]) -> None:
|
|
495
|
+
"""Add a way of reaching an agent. A hosted runtime is a class and this line."""
|
|
496
|
+
_REGISTRY[key] = factory
|
|
497
|
+
|
|
498
|
+
|
|
499
|
+
def resolve(key: str) -> Callable[..., Target]:
|
|
500
|
+
if key not in _REGISTRY:
|
|
501
|
+
raise NotImplementedError(
|
|
502
|
+
f"no target {key!r}; registered targets are {', '.join(sorted(_REGISTRY))}"
|
|
503
|
+
)
|
|
504
|
+
return _REGISTRY[key]
|
|
505
|
+
|
|
506
|
+
|
|
507
|
+
def supported() -> tuple[str, ...]:
|
|
508
|
+
return tuple(sorted(_REGISTRY))
|