agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,506 @@
|
|
|
1
|
+
"""Hosted HTTP chat calls against an already-provisioned Bundle V2 process world."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import os
|
|
7
|
+
import re
|
|
8
|
+
import time
|
|
9
|
+
from datetime import datetime, timezone
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
from typing import Any, Mapping, Sequence
|
|
12
|
+
from urllib.parse import urljoin
|
|
13
|
+
|
|
14
|
+
from fi.simulate.agent.wrapper import AgentInput
|
|
15
|
+
from fi.simulate.agent.wrappers.http import HTTPAgentWrapper
|
|
16
|
+
|
|
17
|
+
from .call_runner import ArtifactUploader, CallRunnerContext
|
|
18
|
+
from .contract import AgentContract
|
|
19
|
+
from .hosted_scheduler import CallAborted, CallOutcome, Scenario, World
|
|
20
|
+
from .outbound import ArtifactKind, format_rfc3339_millis
|
|
21
|
+
from .process_runtime import EnvironmentRuntime
|
|
22
|
+
from .run.conversation import TargetConversationEnded, Transcript, converse
|
|
23
|
+
from .scenario import Scenario as ConversationScenario
|
|
24
|
+
from .world.runtime import Call, GeneratedWorld
|
|
25
|
+
from .world.stores.postgres import AttachedPostgresStore
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
DEFAULT_CHAT_TARGET_TIMEOUT_SECONDS = 120.0
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _chat_target_timeout_seconds() -> float:
|
|
32
|
+
"""Return the per-turn target deadline without accepting unusable values."""
|
|
33
|
+
raw = os.getenv("ALK_CHAT_TARGET_TIMEOUT_SECONDS", "").strip()
|
|
34
|
+
if not raw:
|
|
35
|
+
return DEFAULT_CHAT_TARGET_TIMEOUT_SECONDS
|
|
36
|
+
try:
|
|
37
|
+
configured = float(raw)
|
|
38
|
+
except ValueError:
|
|
39
|
+
return DEFAULT_CHAT_TARGET_TIMEOUT_SECONDS
|
|
40
|
+
return (
|
|
41
|
+
configured
|
|
42
|
+
if 1.0 <= configured <= 600.0
|
|
43
|
+
else DEFAULT_CHAT_TARGET_TIMEOUT_SECONDS
|
|
44
|
+
)
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def _duration_ms(started: datetime, ended: datetime) -> int:
|
|
48
|
+
return max(0, round((ended - started).total_seconds() * 1000))
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def _scenario_document(bundle_dir: Path, key: str) -> dict[str, Any]:
|
|
52
|
+
for path in sorted((bundle_dir / "scenarios").glob("*/scenario.json")):
|
|
53
|
+
try:
|
|
54
|
+
body = json.loads(path.read_text(encoding="utf-8"))
|
|
55
|
+
except (OSError, ValueError):
|
|
56
|
+
continue
|
|
57
|
+
if isinstance(body, dict) and body.get("scenario_key") == key:
|
|
58
|
+
return body
|
|
59
|
+
raise CallAborted(f"chat_scenario_document_unavailable: scenario_key={key!r}")
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def _qmark_to_postgres(statement: str) -> str:
|
|
63
|
+
"""Translate generated SQLite-style positional placeholders, never SQL structure."""
|
|
64
|
+
return re.sub(r"\?", "%s", statement)
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
class _HostedToolStore:
|
|
68
|
+
"""Generated handler ``Db`` adapter over the leased world's attached Postgres database."""
|
|
69
|
+
|
|
70
|
+
key = "hosted-postgres"
|
|
71
|
+
|
|
72
|
+
def __init__(self, dsn: str) -> None:
|
|
73
|
+
self._store = AttachedPostgresStore(dsn)
|
|
74
|
+
|
|
75
|
+
def start(self) -> None:
|
|
76
|
+
self._store.start()
|
|
77
|
+
|
|
78
|
+
def stop(self) -> None:
|
|
79
|
+
return
|
|
80
|
+
|
|
81
|
+
def query(self, sql: str, params: Sequence[Any] = ()) -> list[dict[str, Any]]:
|
|
82
|
+
return self._store.query(_qmark_to_postgres(sql), params)
|
|
83
|
+
|
|
84
|
+
def execute(self, sql: str, params: Sequence[Any] = ()) -> int:
|
|
85
|
+
return self._store.execute(_qmark_to_postgres(sql), params)
|
|
86
|
+
|
|
87
|
+
def state(
|
|
88
|
+
self, only: Sequence[str] | None = None
|
|
89
|
+
) -> dict[str, list[dict[str, Any]]]:
|
|
90
|
+
return self._store.state(only=only)
|
|
91
|
+
|
|
92
|
+
def collections(self) -> list[str]:
|
|
93
|
+
return list(self._store.state())
|
|
94
|
+
|
|
95
|
+
def records(self, collection: str) -> list[dict[str, Any]]:
|
|
96
|
+
return self._store.table(collection)
|
|
97
|
+
|
|
98
|
+
def add(self, collection: str, record: Mapping[str, Any]) -> dict[str, Any]:
|
|
99
|
+
return self._store.add(collection, dict(record))
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def _tool_world(
|
|
103
|
+
bundle_dir: Path,
|
|
104
|
+
contract: AgentContract,
|
|
105
|
+
runtime: EnvironmentRuntime,
|
|
106
|
+
source_directory: Path | None = None,
|
|
107
|
+
) -> GeneratedWorld:
|
|
108
|
+
endpoint = runtime.endpoints.get("world_db")
|
|
109
|
+
if endpoint is None or endpoint.protocol != "postgres":
|
|
110
|
+
raise CallAborted(
|
|
111
|
+
"chat_world_unavailable: world_db postgres endpoint is absent"
|
|
112
|
+
)
|
|
113
|
+
world = GeneratedWorld(store=_HostedToolStore(endpoint.address))
|
|
114
|
+
world.tools = [tool.model_dump(mode="json") for tool in contract.tools]
|
|
115
|
+
world.handlers = {}
|
|
116
|
+
for tool in contract.tools:
|
|
117
|
+
path = bundle_dir / "handlers" / f"{tool.name}.py"
|
|
118
|
+
if path.is_file():
|
|
119
|
+
world.handlers[tool.name] = path.read_text(encoding="utf-8")
|
|
120
|
+
world.refusal_signature = contract.refusal_signature
|
|
121
|
+
if source_directory is not None:
|
|
122
|
+
world.reach(str(source_directory))
|
|
123
|
+
return world
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def _tools(contract: AgentContract) -> list[dict[str, Any]]:
|
|
127
|
+
values: list[dict[str, Any]] = []
|
|
128
|
+
for spec in contract.tools:
|
|
129
|
+
properties = {
|
|
130
|
+
argument: {"type": _json_type(spec.arg_types.get(argument, "string"))}
|
|
131
|
+
for argument in spec.args
|
|
132
|
+
}
|
|
133
|
+
values.append(
|
|
134
|
+
{
|
|
135
|
+
"name": spec.name,
|
|
136
|
+
"description": spec.description,
|
|
137
|
+
"parameters": {
|
|
138
|
+
"type": "object",
|
|
139
|
+
"properties": properties,
|
|
140
|
+
"required": list(spec.args),
|
|
141
|
+
},
|
|
142
|
+
}
|
|
143
|
+
)
|
|
144
|
+
return values
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def _json_type(declared: str) -> str:
|
|
148
|
+
normalized = str(declared or "").lower()
|
|
149
|
+
if any(mark in normalized for mark in ("int", "float", "number")):
|
|
150
|
+
return "number"
|
|
151
|
+
if "bool" in normalized:
|
|
152
|
+
return "boolean"
|
|
153
|
+
if any(mark in normalized for mark in ("list", "array", "sequence")):
|
|
154
|
+
return "array"
|
|
155
|
+
if any(mark in normalized for mark in ("dict", "map", "object")):
|
|
156
|
+
return "object"
|
|
157
|
+
return "string"
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def _tool_call(call: dict[str, Any], index: int) -> tuple[str, dict[str, Any], str]:
|
|
161
|
+
function = call.get("function") if isinstance(call.get("function"), dict) else {}
|
|
162
|
+
name = str(call.get("name") or function.get("name") or "")
|
|
163
|
+
raw = call.get("arguments", function.get("arguments", {}))
|
|
164
|
+
if isinstance(raw, str):
|
|
165
|
+
try:
|
|
166
|
+
arguments = json.loads(raw)
|
|
167
|
+
except ValueError:
|
|
168
|
+
arguments = {"_raw": raw}
|
|
169
|
+
else:
|
|
170
|
+
arguments = dict(raw or {}) if isinstance(raw, dict) else {}
|
|
171
|
+
return name, arguments, str(call.get("id") or f"call_{index}")
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def _tool_response_result(response: Mapping[str, Any]) -> Any:
|
|
175
|
+
"""Return the callback's real tool result without inventing a second execution."""
|
|
176
|
+
value = response.get("result", response.get("content"))
|
|
177
|
+
if isinstance(value, str):
|
|
178
|
+
try:
|
|
179
|
+
return json.loads(value)
|
|
180
|
+
except ValueError:
|
|
181
|
+
return value
|
|
182
|
+
return value
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
def _record_completed_tool_call(
|
|
186
|
+
world: GeneratedWorld,
|
|
187
|
+
*,
|
|
188
|
+
name: str,
|
|
189
|
+
arguments: Mapping[str, Any],
|
|
190
|
+
response: Mapping[str, Any],
|
|
191
|
+
) -> None:
|
|
192
|
+
"""Record a tool the submitted callback already executed.
|
|
193
|
+
|
|
194
|
+
Callback-backed agents can return the request and its completed response together. Replaying
|
|
195
|
+
that request through the generated world both risks repeating a side effect and incorrectly
|
|
196
|
+
turns a real success into ``no such tool`` when no mock handler was authored. The callback's
|
|
197
|
+
response is the authoritative execution evidence at this seam.
|
|
198
|
+
"""
|
|
199
|
+
error_value = response.get("error")
|
|
200
|
+
error = str(error_value) if error_value not in (None, "") else ""
|
|
201
|
+
refused = bool(response.get("refused", False))
|
|
202
|
+
declared_success = response.get("success", response.get("ok"))
|
|
203
|
+
ok = (
|
|
204
|
+
bool(declared_success)
|
|
205
|
+
if declared_success is not None
|
|
206
|
+
else not error and not refused
|
|
207
|
+
)
|
|
208
|
+
world.calls.append(
|
|
209
|
+
Call(
|
|
210
|
+
name=name,
|
|
211
|
+
arguments=dict(arguments),
|
|
212
|
+
result=_tool_response_result(response),
|
|
213
|
+
ok=ok,
|
|
214
|
+
refused=refused,
|
|
215
|
+
error=error,
|
|
216
|
+
at=time.time(),
|
|
217
|
+
)
|
|
218
|
+
)
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
class _HostedChatTarget:
|
|
222
|
+
"""The already-running Bundle V2 chat process, exposed as the normal conversation target.
|
|
223
|
+
|
|
224
|
+
``converse`` owns the simulated customer's turns. This target owns only the submitted
|
|
225
|
+
agent's side of the exchange and response-carried tool evidence. Keeping that split identical
|
|
226
|
+
to the local repository target prevents the hosted lane from silently becoming a one-message
|
|
227
|
+
smoke test again.
|
|
228
|
+
"""
|
|
229
|
+
|
|
230
|
+
key = "hosted_repository"
|
|
231
|
+
|
|
232
|
+
def __init__(
|
|
233
|
+
self,
|
|
234
|
+
*,
|
|
235
|
+
wrapper: Any,
|
|
236
|
+
contract: AgentContract,
|
|
237
|
+
world: GeneratedWorld,
|
|
238
|
+
scenario_key: str,
|
|
239
|
+
scenario_id: str,
|
|
240
|
+
) -> None:
|
|
241
|
+
self._wrapper = wrapper
|
|
242
|
+
self._contract = contract
|
|
243
|
+
self.world = world
|
|
244
|
+
self._scenario_key = scenario_key
|
|
245
|
+
self._scenario_id = scenario_id
|
|
246
|
+
self._messages: list[dict[str, Any]] = []
|
|
247
|
+
self._turn = 0
|
|
248
|
+
|
|
249
|
+
async def open(self) -> None:
|
|
250
|
+
return
|
|
251
|
+
|
|
252
|
+
async def say(self, utterance: str) -> str:
|
|
253
|
+
self._messages.append({"role": "user", "content": utterance})
|
|
254
|
+
for continuation in range(8):
|
|
255
|
+
response = await self._wrapper.call(
|
|
256
|
+
AgentInput(
|
|
257
|
+
thread_id=self._scenario_key,
|
|
258
|
+
execution_id=self._scenario_id,
|
|
259
|
+
turn_index=self._turn,
|
|
260
|
+
scenario_name=self._scenario_key,
|
|
261
|
+
modality="text",
|
|
262
|
+
messages=list(self._messages),
|
|
263
|
+
new_message=dict(self._messages[-1]),
|
|
264
|
+
tools=_tools(self._contract)
|
|
265
|
+
if self._contract.runtime
|
|
266
|
+
and self._contract.runtime.interface
|
|
267
|
+
and self._contract.runtime.interface.include_tools
|
|
268
|
+
else [],
|
|
269
|
+
)
|
|
270
|
+
)
|
|
271
|
+
trace = dict((response.metadata or {}).get("external_agent") or {})
|
|
272
|
+
if trace and not trace.get("success", False):
|
|
273
|
+
raise RuntimeError(
|
|
274
|
+
str(trace.get("error") or "submitted endpoint request failed")
|
|
275
|
+
)
|
|
276
|
+
conversation_ended = bool(
|
|
277
|
+
(response.metadata or {}).get("conversation_ended")
|
|
278
|
+
)
|
|
279
|
+
returned = list(response.tool_calls or [])
|
|
280
|
+
if not returned:
|
|
281
|
+
answer = response.content.strip()
|
|
282
|
+
if conversation_ended:
|
|
283
|
+
raise TargetConversationEnded(answer)
|
|
284
|
+
self._messages.append({"role": "assistant", "content": answer})
|
|
285
|
+
self._turn += 1
|
|
286
|
+
return answer
|
|
287
|
+
|
|
288
|
+
self._messages.append(
|
|
289
|
+
{
|
|
290
|
+
"role": "assistant",
|
|
291
|
+
"content": response.content or "",
|
|
292
|
+
"tool_calls": returned,
|
|
293
|
+
}
|
|
294
|
+
)
|
|
295
|
+
response_by_id = {
|
|
296
|
+
str(item.get("tool_call_id") or item.get("id") or ""): item
|
|
297
|
+
for item in response.tool_responses or []
|
|
298
|
+
if isinstance(item, Mapping)
|
|
299
|
+
and (item.get("tool_call_id") or item.get("id"))
|
|
300
|
+
}
|
|
301
|
+
returned_ids: set[str] = set()
|
|
302
|
+
for index, call in enumerate(returned, start=1):
|
|
303
|
+
name, arguments, call_id = _tool_call(call, index)
|
|
304
|
+
returned_ids.add(call_id)
|
|
305
|
+
provided = response_by_id.get(call_id)
|
|
306
|
+
if provided is not None:
|
|
307
|
+
_record_completed_tool_call(
|
|
308
|
+
self.world,
|
|
309
|
+
name=name,
|
|
310
|
+
arguments=arguments,
|
|
311
|
+
response=provided,
|
|
312
|
+
)
|
|
313
|
+
result_content = provided.get("content", provided.get("result"))
|
|
314
|
+
if not isinstance(result_content, str):
|
|
315
|
+
result_content = json.dumps(result_content, default=str)
|
|
316
|
+
else:
|
|
317
|
+
result = self.world.handle_tool_call(
|
|
318
|
+
{"id": call_id, "name": name, "arguments": arguments}
|
|
319
|
+
)
|
|
320
|
+
result_content = (
|
|
321
|
+
result.content if result is not None else f"no such tool {name}"
|
|
322
|
+
)
|
|
323
|
+
self._messages.append(
|
|
324
|
+
{
|
|
325
|
+
"role": "tool",
|
|
326
|
+
"tool_call_id": call_id,
|
|
327
|
+
"name": name,
|
|
328
|
+
"content": result_content,
|
|
329
|
+
}
|
|
330
|
+
)
|
|
331
|
+
|
|
332
|
+
# A callback-backed repository has already run its own real tools. It returns their
|
|
333
|
+
# responses beside the final text; replaying the calls above records deterministic
|
|
334
|
+
# evidence in the generated world, but asking the callback a second time would run the
|
|
335
|
+
# tools twice. HTTP agents that only return requests continue normally with the
|
|
336
|
+
# generated-world responses appended above.
|
|
337
|
+
provided_ids = set(response_by_id)
|
|
338
|
+
if returned_ids and returned_ids.issubset(provided_ids):
|
|
339
|
+
answer = response.content.strip()
|
|
340
|
+
if conversation_ended:
|
|
341
|
+
raise TargetConversationEnded(answer)
|
|
342
|
+
self._messages.append({"role": "assistant", "content": answer})
|
|
343
|
+
self._turn += 1
|
|
344
|
+
return answer
|
|
345
|
+
raise RuntimeError(
|
|
346
|
+
"submitted chat agent exceeded 8 tool continuations in one turn"
|
|
347
|
+
)
|
|
348
|
+
|
|
349
|
+
async def close(self) -> None:
|
|
350
|
+
return
|
|
351
|
+
|
|
352
|
+
@property
|
|
353
|
+
def spent_usd(self) -> float:
|
|
354
|
+
# Provider-side target cost is not observable at this transport seam.
|
|
355
|
+
return 0.0
|
|
356
|
+
|
|
357
|
+
|
|
358
|
+
def _conversation_scenario(document: dict[str, Any]) -> ConversationScenario:
|
|
359
|
+
try:
|
|
360
|
+
normalized = dict(document)
|
|
361
|
+
normalized.setdefault(
|
|
362
|
+
"name", str(normalized.get("scenario_key") or "hosted-chat-scenario")
|
|
363
|
+
)
|
|
364
|
+
return ConversationScenario.model_validate(normalized)
|
|
365
|
+
except Exception as exc: # noqa: BLE001 - normalize malformed bundle content at the call seam
|
|
366
|
+
raise CallAborted(f"chat_scenario_invalid: {exc}") from exc
|
|
367
|
+
|
|
368
|
+
|
|
369
|
+
async def _drive_conversation(
|
|
370
|
+
target: _HostedChatTarget,
|
|
371
|
+
scenario: ConversationScenario,
|
|
372
|
+
contract: AgentContract,
|
|
373
|
+
bundle_dir: Path,
|
|
374
|
+
) -> Transcript:
|
|
375
|
+
return await converse(
|
|
376
|
+
target,
|
|
377
|
+
scenario,
|
|
378
|
+
contract,
|
|
379
|
+
world_root=bundle_dir,
|
|
380
|
+
)
|
|
381
|
+
|
|
382
|
+
|
|
383
|
+
class HostedChatCallRunner:
|
|
384
|
+
"""Drive a repository chat ingress inside its leased world."""
|
|
385
|
+
|
|
386
|
+
def __init__(self, adapter: ArtifactUploader, context: CallRunnerContext) -> None:
|
|
387
|
+
self._adapter = adapter
|
|
388
|
+
self._context = context
|
|
389
|
+
contract_path = context.bundle_dir / "contract.json"
|
|
390
|
+
if not contract_path.is_file():
|
|
391
|
+
self._contract: AgentContract | None = None
|
|
392
|
+
else:
|
|
393
|
+
self._contract = AgentContract.model_validate_json(
|
|
394
|
+
contract_path.read_text(encoding="utf-8")
|
|
395
|
+
)
|
|
396
|
+
|
|
397
|
+
async def run(
|
|
398
|
+
self,
|
|
399
|
+
scenario: Scenario,
|
|
400
|
+
runtime: EnvironmentRuntime,
|
|
401
|
+
*,
|
|
402
|
+
world: World | None = None,
|
|
403
|
+
) -> CallOutcome:
|
|
404
|
+
del (
|
|
405
|
+
world
|
|
406
|
+
) # Handler execution uses the same leased world's endpoint from runtime.
|
|
407
|
+
if self._contract is None:
|
|
408
|
+
raise CallAborted(
|
|
409
|
+
"chat_contract_unavailable: bundle/contract.json is absent"
|
|
410
|
+
)
|
|
411
|
+
interface = self._contract.runtime.interface if self._contract.runtime else None
|
|
412
|
+
if interface is None or interface.kind not in {"http", "callable"}:
|
|
413
|
+
raise CallAborted(
|
|
414
|
+
"chat_interface_unsupported: an HTTP or callable runtime interface is required"
|
|
415
|
+
)
|
|
416
|
+
endpoint = runtime.endpoints.get("target_http")
|
|
417
|
+
if endpoint is None:
|
|
418
|
+
raise CallAborted(
|
|
419
|
+
"chat_capability_unavailable: target_http endpoint is absent"
|
|
420
|
+
)
|
|
421
|
+
|
|
422
|
+
document = _scenario_document(self._context.bundle_dir, scenario.scenario_key)
|
|
423
|
+
conversation_scenario = _conversation_scenario(document)
|
|
424
|
+
if not conversation_scenario.instruction.strip():
|
|
425
|
+
raise CallAborted("chat_scenario_invalid: instruction is empty")
|
|
426
|
+
target_world = _tool_world(
|
|
427
|
+
self._context.bundle_dir,
|
|
428
|
+
self._contract,
|
|
429
|
+
runtime,
|
|
430
|
+
self._context.source_directory,
|
|
431
|
+
)
|
|
432
|
+
adapter_path = "/invoke" if interface.kind == "callable" else interface.path
|
|
433
|
+
adapter_protocol = (
|
|
434
|
+
"fi.alk" if interface.kind == "callable" else interface.protocol
|
|
435
|
+
)
|
|
436
|
+
wrapper = HTTPAgentWrapper(
|
|
437
|
+
endpoint=urljoin(
|
|
438
|
+
endpoint.address.rstrip("/") + "/", adapter_path.lstrip("/")
|
|
439
|
+
),
|
|
440
|
+
protocol=adapter_protocol,
|
|
441
|
+
include_tools=interface.include_tools,
|
|
442
|
+
timeout=_chat_target_timeout_seconds(),
|
|
443
|
+
metadata={
|
|
444
|
+
"target": "hosted_repository_runtime",
|
|
445
|
+
"scenario": scenario.scenario_key,
|
|
446
|
+
},
|
|
447
|
+
)
|
|
448
|
+
started = datetime.now(timezone.utc)
|
|
449
|
+
try:
|
|
450
|
+
transcript = await _drive_conversation(
|
|
451
|
+
_HostedChatTarget(
|
|
452
|
+
wrapper=wrapper,
|
|
453
|
+
contract=self._contract,
|
|
454
|
+
world=target_world,
|
|
455
|
+
scenario_key=scenario.scenario_key,
|
|
456
|
+
scenario_id=scenario.scenario_id,
|
|
457
|
+
),
|
|
458
|
+
conversation_scenario,
|
|
459
|
+
self._contract,
|
|
460
|
+
self._context.bundle_dir,
|
|
461
|
+
)
|
|
462
|
+
except CallAborted:
|
|
463
|
+
raise
|
|
464
|
+
except Exception as exc: # noqa: BLE001 - convert target transport failures to call faults
|
|
465
|
+
raise CallAborted(
|
|
466
|
+
f"chat_target_failed: {type(exc).__name__}: {exc}"
|
|
467
|
+
) from exc
|
|
468
|
+
|
|
469
|
+
ended = datetime.now(timezone.utc)
|
|
470
|
+
transcript_id = await self._adapter.upload_artifact(
|
|
471
|
+
transcript.artifact(),
|
|
472
|
+
kind=ArtifactKind.TRANSCRIPT,
|
|
473
|
+
scenario_key=scenario.scenario_key,
|
|
474
|
+
)
|
|
475
|
+
calls = tuple(transcript.calls)
|
|
476
|
+
if calls:
|
|
477
|
+
tool_trace = "\n".join(
|
|
478
|
+
json.dumps(
|
|
479
|
+
{
|
|
480
|
+
"name": call.name,
|
|
481
|
+
"arguments": call.arguments,
|
|
482
|
+
"result": call.result,
|
|
483
|
+
"ok": call.ok,
|
|
484
|
+
"error": call.error,
|
|
485
|
+
"refused": call.refused,
|
|
486
|
+
"at": call.at,
|
|
487
|
+
},
|
|
488
|
+
sort_keys=True,
|
|
489
|
+
default=str,
|
|
490
|
+
)
|
|
491
|
+
for call in calls
|
|
492
|
+
).encode("utf-8")
|
|
493
|
+
await self._adapter.upload_artifact(
|
|
494
|
+
tool_trace,
|
|
495
|
+
kind=ArtifactKind.TOOL_TRACE,
|
|
496
|
+
scenario_key=scenario.scenario_key,
|
|
497
|
+
)
|
|
498
|
+
return CallOutcome(
|
|
499
|
+
calls=calls,
|
|
500
|
+
turns=len(transcript.exchanges),
|
|
501
|
+
started_at=format_rfc3339_millis(started),
|
|
502
|
+
ended_at=format_rfc3339_millis(ended),
|
|
503
|
+
duration_ms=_duration_ms(started, ended),
|
|
504
|
+
transcript_artifact=transcript_id,
|
|
505
|
+
messages=tuple(transcript.canonical_messages()),
|
|
506
|
+
)
|
fi/alk/harness/checks.py
ADDED
|
@@ -0,0 +1,136 @@
|
|
|
1
|
+
"""Running a check the harness wrote, and deciding what its answer means.
|
|
2
|
+
|
|
3
|
+
A check is Python because an environment can be a database, a filesystem or a page, and any
|
|
4
|
+
little assertion language invented here would fit only the first. It is given the two things a
|
|
5
|
+
run leaves behind and returns a sentence when something is wrong:
|
|
6
|
+
|
|
7
|
+
def check(world, calls):
|
|
8
|
+
rows = world.state()["orders"]
|
|
9
|
+
if len(rows) != 1:
|
|
10
|
+
return f"{len(rows)} orders, expected 1"
|
|
11
|
+
if not any(c.name == "order_combo_meal" for c in calls):
|
|
12
|
+
return "the combo was never ordered"
|
|
13
|
+
return None
|
|
14
|
+
|
|
15
|
+
``world`` is the environment afterwards. ``calls`` is every tool call that was made, each with
|
|
16
|
+
its arguments and whether it succeeded — so a check can insist not only that a call happened but
|
|
17
|
+
that it happened with the right arguments, which is the difference between booking 11 PM and
|
|
18
|
+
booking 10 PM.
|
|
19
|
+
|
|
20
|
+
A check that raises is a broken check, not a failed one, and is reported that way. Confusing the
|
|
21
|
+
two would let a typo read as a finding about the agent.
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
from __future__ import annotations
|
|
25
|
+
|
|
26
|
+
from dataclasses import dataclass
|
|
27
|
+
from typing import Any, Sequence
|
|
28
|
+
|
|
29
|
+
from .world.runtime import Call, GeneratedWorld
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
@dataclass
|
|
33
|
+
class Outcome:
|
|
34
|
+
"""What one check said."""
|
|
35
|
+
|
|
36
|
+
name: str
|
|
37
|
+
held: bool
|
|
38
|
+
said: str = ""
|
|
39
|
+
broken: bool = False
|
|
40
|
+
|
|
41
|
+
def line(self) -> str:
|
|
42
|
+
mark = "!" if self.broken else ("x" if self.held else " ")
|
|
43
|
+
return f" [{mark}] {self.name}" + (f" — {self.said}" if self.said else "")
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def run_check(
|
|
47
|
+
source: str, world: GeneratedWorld, calls: Sequence[Call], *, name: str = "check"
|
|
48
|
+
) -> Outcome:
|
|
49
|
+
"""Execute one check against what the run left behind."""
|
|
50
|
+
namespace: dict[str, Any] = {}
|
|
51
|
+
try:
|
|
52
|
+
exec(compile(source, f"<check:{name}>", "exec"), namespace)
|
|
53
|
+
except Exception as failed:
|
|
54
|
+
return Outcome(
|
|
55
|
+
name, False, f"the check would not compile: {failed}", broken=True
|
|
56
|
+
)
|
|
57
|
+
|
|
58
|
+
checker = namespace.get("check")
|
|
59
|
+
if not callable(checker):
|
|
60
|
+
return Outcome(
|
|
61
|
+
name, False, "the check defines no check(world, calls)", broken=True
|
|
62
|
+
)
|
|
63
|
+
|
|
64
|
+
try:
|
|
65
|
+
said = checker(world, list(calls))
|
|
66
|
+
except Exception as failed:
|
|
67
|
+
# The check is at fault, not the agent. A KeyError in an assertion is our bug, and
|
|
68
|
+
# scoring it against the agent is how a harness invents findings.
|
|
69
|
+
return Outcome(
|
|
70
|
+
name,
|
|
71
|
+
False,
|
|
72
|
+
f"the check raised {type(failed).__name__}: {str(failed)[:200]}",
|
|
73
|
+
broken=True,
|
|
74
|
+
)
|
|
75
|
+
|
|
76
|
+
if said is None or said is True or (isinstance(said, str) and not said.strip()):
|
|
77
|
+
return Outcome(name, True)
|
|
78
|
+
if said is False:
|
|
79
|
+
return Outcome(name, False, "False")
|
|
80
|
+
if not isinstance(said, str):
|
|
81
|
+
return Outcome(
|
|
82
|
+
name,
|
|
83
|
+
False,
|
|
84
|
+
f"the check returned {type(said).__name__} {repr(said)[:200]}; a check returns a "
|
|
85
|
+
"sentence naming what is wrong, or None when it held.",
|
|
86
|
+
broken=True,
|
|
87
|
+
)
|
|
88
|
+
return Outcome(name, False, said)
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def all_held(outcomes: Sequence[Outcome]) -> bool:
|
|
92
|
+
return all(one.held for one in outcomes) and not any(one.broken for one in outcomes)
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def broken(outcomes: Sequence[Outcome]) -> list[Outcome]:
|
|
96
|
+
return [one for one in outcomes if one.broken]
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def run_world_check(
|
|
100
|
+
source: str, world: GeneratedWorld, *, name: str = "check"
|
|
101
|
+
) -> Outcome:
|
|
102
|
+
"""Execute one check about the world itself, rather than about a run.
|
|
103
|
+
|
|
104
|
+
A world check asks whether the environment is usable at all, so it is written ``check(world)``
|
|
105
|
+
and there are no calls to give it. Both arities are accepted, because the difference is not
|
|
106
|
+
worth a rejection: a check written ``check(world, calls)`` out of habit is answering the same
|
|
107
|
+
question, and gets an empty list.
|
|
108
|
+
"""
|
|
109
|
+
import inspect
|
|
110
|
+
|
|
111
|
+
namespace: dict[str, Any] = {}
|
|
112
|
+
try:
|
|
113
|
+
exec(compile(source, f"<world-check:{name}>", "exec"), namespace)
|
|
114
|
+
except Exception as failed:
|
|
115
|
+
return Outcome(
|
|
116
|
+
name, False, f"the check would not compile: {failed}", broken=True
|
|
117
|
+
)
|
|
118
|
+
|
|
119
|
+
checker = namespace.get("check")
|
|
120
|
+
if not callable(checker):
|
|
121
|
+
return Outcome(name, False, "the check defines no check(world)", broken=True)
|
|
122
|
+
|
|
123
|
+
try:
|
|
124
|
+
wants = len(inspect.signature(checker).parameters)
|
|
125
|
+
except (TypeError, ValueError):
|
|
126
|
+
wants = 1
|
|
127
|
+
try:
|
|
128
|
+
said = checker(world) if wants < 2 else checker(world, [])
|
|
129
|
+
except Exception as failed:
|
|
130
|
+
return Outcome(
|
|
131
|
+
name,
|
|
132
|
+
False,
|
|
133
|
+
f"the check raised {type(failed).__name__}: {str(failed)[:200]}",
|
|
134
|
+
broken=True,
|
|
135
|
+
)
|
|
136
|
+
return Outcome(name, said is None, "" if said is None else str(said))
|