agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,169 @@
|
|
|
1
|
+
"""Stage one: read an agent and produce its contract.
|
|
2
|
+
|
|
3
|
+
The stage is the same whatever the agent is. What changes between a repository, a provider
|
|
4
|
+
connection and a pasted definition is where the truth lives, and that comes from the source.
|
|
5
|
+
|
|
6
|
+
It stays open after the first answer, because a contract is usually right on the second look and
|
|
7
|
+
not the first. Correcting it is the next thing said, not a re-run.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import json
|
|
13
|
+
import os
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
from typing import Any, Callable
|
|
16
|
+
|
|
17
|
+
from .config import artifact_dir, load_skill, read_only_session
|
|
18
|
+
from .contract import AgentContract
|
|
19
|
+
from .session import Stage
|
|
20
|
+
from .sources import AgentSource
|
|
21
|
+
from .tools import CONTRACT_SERVER, contract_tools
|
|
22
|
+
|
|
23
|
+
SKILL = "understand-agent"
|
|
24
|
+
PROVIDER_IMPORT_PROFILE_PATH_ENV = "ALK_PROVIDER_IMPORT_PROFILE_PATH"
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _provider_import_briefing() -> str:
|
|
28
|
+
"""Load the sanitized external-provider definition prepared by the control process."""
|
|
29
|
+
configured = os.environ.get(PROVIDER_IMPORT_PROFILE_PATH_ENV, "").strip()
|
|
30
|
+
if not configured:
|
|
31
|
+
return ""
|
|
32
|
+
path = Path(configured)
|
|
33
|
+
try:
|
|
34
|
+
profile = json.loads(path.read_text(encoding="utf-8"))
|
|
35
|
+
except (OSError, ValueError):
|
|
36
|
+
return ""
|
|
37
|
+
return (
|
|
38
|
+
"\n\n## Imported provider target (authoritative, read-only)\n\n"
|
|
39
|
+
"The submitted repository implements this target's environment-facing webhooks. "
|
|
40
|
+
"The externally hosted assistant definition below is equally authoritative for its "
|
|
41
|
+
"conversation, prompt, model, voice, and tool schemas. Reconcile both sources. Do not "
|
|
42
|
+
"invent behavior or tool inputs that conflict with the provider definition.\n\n"
|
|
43
|
+
f"```json\n{json.dumps(profile, indent=2, sort_keys=True)}\n```"
|
|
44
|
+
)
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def _eval_catalogue_briefing(available_evals: list[dict[str, Any]] | None) -> str:
|
|
48
|
+
"""The eval catalogue grouped by modality, since the contract has not claimed one yet."""
|
|
49
|
+
grouped: dict[str, list[str]] = {}
|
|
50
|
+
for one in available_evals or []:
|
|
51
|
+
if not isinstance(one, dict):
|
|
52
|
+
continue
|
|
53
|
+
name = str(one.get("name") or "").strip()
|
|
54
|
+
if not name:
|
|
55
|
+
continue
|
|
56
|
+
keys = ", ".join(str(key) for key in one.get("required_keys") or [])
|
|
57
|
+
grouped.setdefault(
|
|
58
|
+
str(one.get("modality") or "any").strip().lower(), []
|
|
59
|
+
).append(
|
|
60
|
+
"- {name}: {description}{keys}".format(
|
|
61
|
+
name=name,
|
|
62
|
+
description=str(one.get("description") or "").strip()[:240],
|
|
63
|
+
keys=f" [needs: {keys}]" if keys else "",
|
|
64
|
+
)
|
|
65
|
+
)
|
|
66
|
+
if not grouped:
|
|
67
|
+
return ""
|
|
68
|
+
sections = [
|
|
69
|
+
f"### Applies to {kind}\n\n" + "\n".join(sorted(lines))
|
|
70
|
+
for kind, lines in sorted(grouped.items())
|
|
71
|
+
]
|
|
72
|
+
return (
|
|
73
|
+
"\n\n## Evals this platform can run on the finished calls\n\n"
|
|
74
|
+
"Record the ones this agent should be judged by in `chosen_evals`, by exact name. Each one "
|
|
75
|
+
"runs as a judge on every call of every scenario, so choose the few that tell you "
|
|
76
|
+
"something the scenarios' own deterministic checks cannot, and choose none rather than "
|
|
77
|
+
"padding.\n\n"
|
|
78
|
+
"Choose only from the section matching the `modality` you are about to record. The "
|
|
79
|
+
"platform refuses an eval belonging to another modality, so an eval for spoken calls "
|
|
80
|
+
"recorded against a chat agent costs the whole submission.\n\n"
|
|
81
|
+
"Two ways to choose wrongly, both common. An eval whose subject only exists in speech, "
|
|
82
|
+
"dead air, voicemail detection, voicemail handling, is meaningless for a chat agent and "
|
|
83
|
+
"worth having for a voice one. An eval named for a domain, misselling, advice authority, "
|
|
84
|
+
"lead qualification, claim intake, intake field accuracy, is only worth choosing when this "
|
|
85
|
+
"agent's own tools and constraints show it doing that work; choose it from the evidence in "
|
|
86
|
+
"front of you, never because its name sounds close to the agent's industry.\n\n"
|
|
87
|
+
+ "\n\n".join(sections)
|
|
88
|
+
)
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def open_stage(
|
|
92
|
+
source: AgentSource,
|
|
93
|
+
*,
|
|
94
|
+
out: Path | None = None,
|
|
95
|
+
ask: Callable[..., Any] | None = None,
|
|
96
|
+
max_turns: int = 70,
|
|
97
|
+
available_evals: list[dict[str, Any]] | None = None,
|
|
98
|
+
) -> tuple[Stage, Path]:
|
|
99
|
+
"""A live understand-the-agent stage, and where it will write."""
|
|
100
|
+
destination = out or artifact_dir(source.name)
|
|
101
|
+
spec = read_only_session(
|
|
102
|
+
system_prompt=(
|
|
103
|
+
f"{load_skill(SKILL)}\n\n## This agent\n\n{source.briefing()}"
|
|
104
|
+
# A source-free connect-only target already carries this definition in its
|
|
105
|
+
# ProviderSource briefing. Repository-backed provider imports still need the
|
|
106
|
+
# separately injected profile so the model can reconcile both sources of truth.
|
|
107
|
+
f"{_provider_import_briefing() if source.kind != 'provider' else ''}"
|
|
108
|
+
f"{_eval_catalogue_briefing(available_evals)}"
|
|
109
|
+
),
|
|
110
|
+
cwd=source.workdir(),
|
|
111
|
+
servers={
|
|
112
|
+
**source.servers(),
|
|
113
|
+
CONTRACT_SERVER: contract_tools(
|
|
114
|
+
destination,
|
|
115
|
+
available_evals,
|
|
116
|
+
source_root=source.workdir() if source.kind == "repo" else None,
|
|
117
|
+
),
|
|
118
|
+
},
|
|
119
|
+
extra_builtins=source.builtin_tools(),
|
|
120
|
+
max_turns=max_turns,
|
|
121
|
+
)
|
|
122
|
+
if ask is not None:
|
|
123
|
+
spec.permission_override = ask
|
|
124
|
+
return Stage(spec, name=SKILL), destination
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def opening(source: AgentSource) -> str:
|
|
128
|
+
# The name is only a label for the artifact folder, and saying so matters: told to "read
|
|
129
|
+
# the agent named verify_fix", a model went hunting the whole workspace for something
|
|
130
|
+
# called verify_fix instead of reading the path it was given.
|
|
131
|
+
return (
|
|
132
|
+
"Read this agent and produce its contract. Where it lives is in your briefing; "
|
|
133
|
+
f"{source.name!r} is only the label its artifacts are filed under, not something to "
|
|
134
|
+
"search for.\n\n"
|
|
135
|
+
"Work through the tools, their exact argument names and types, the constrained argument "
|
|
136
|
+
"values, the rules it enforces, and its data. Ask me if the source genuinely does not "
|
|
137
|
+
"settle something that changes what gets built. Call submit_contract when you are done."
|
|
138
|
+
)
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def load(destination: Path) -> AgentContract | None:
|
|
142
|
+
"""The contract on disk, if the stage produced one."""
|
|
143
|
+
path = Path(destination) / "contract.json"
|
|
144
|
+
if not path.exists():
|
|
145
|
+
return None
|
|
146
|
+
return AgentContract.model_validate(json.loads(path.read_text(encoding="utf-8")))
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
async def understand(
|
|
150
|
+
source: AgentSource,
|
|
151
|
+
*,
|
|
152
|
+
out: Path | None = None,
|
|
153
|
+
follow_ups: list[str] | None = None,
|
|
154
|
+
on_event: Callable[..., Any] | None = None,
|
|
155
|
+
ask: Callable[..., Any] | None = None,
|
|
156
|
+
max_turns: int = 70,
|
|
157
|
+
) -> AgentContract | None:
|
|
158
|
+
"""Run the stage start to finish and return the contract.
|
|
159
|
+
|
|
160
|
+
``follow_ups`` are corrections applied in the same session, the scripted equivalent of an
|
|
161
|
+
operator typing them. ``ask`` handles clarifying questions; without it the model records what
|
|
162
|
+
it could not resolve in ``open_questions`` instead of blocking.
|
|
163
|
+
"""
|
|
164
|
+
stage, destination = open_stage(source, out=out, ask=ask, max_turns=max_turns)
|
|
165
|
+
async with stage:
|
|
166
|
+
await stage.say(opening(source), on_event=on_event)
|
|
167
|
+
for follow_up in follow_ups or []:
|
|
168
|
+
await stage.say(follow_up, on_event=on_event)
|
|
169
|
+
return load(destination)
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
"""Which recorded mailbox greeting, if any, a voicemail scenario is heard through.
|
|
2
|
+
|
|
3
|
+
Clips must greet without naming anybody, so one recording fits any persona. The catalogue is a
|
|
4
|
+
local file and is not committed: absent, the mailbox speaks its greeting and the tone is generated.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import json
|
|
10
|
+
import os
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
from typing import Any
|
|
13
|
+
|
|
14
|
+
CATALOG = Path(__file__).with_name("data") / "voicemail" / "catalog.json"
|
|
15
|
+
# A run may point somewhere else, so a deployment can serve these from object storage.
|
|
16
|
+
CATALOG_ENV = "ALK_VOICEMAIL_CATALOG"
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def _entries() -> list[dict[str, Any]]:
|
|
20
|
+
path = Path(os.environ.get(CATALOG_ENV, "").strip() or CATALOG)
|
|
21
|
+
try:
|
|
22
|
+
body = json.loads(path.read_text(encoding="utf-8"))
|
|
23
|
+
except Exception: # noqa: BLE001 - no catalogue is the ordinary case, never an error
|
|
24
|
+
return []
|
|
25
|
+
return [one for one in body if isinstance(one, dict)] if isinstance(body, list) else []
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _resolved(entry: dict[str, Any]) -> str:
|
|
29
|
+
"""Where the audio actually is: a URL if the catalogue serves one, else a file beside us."""
|
|
30
|
+
url = str(entry.get("url") or "").strip()
|
|
31
|
+
if url:
|
|
32
|
+
return url
|
|
33
|
+
raw = str(entry.get("path") or "").strip()
|
|
34
|
+
if not raw:
|
|
35
|
+
return ""
|
|
36
|
+
here = Path(raw)
|
|
37
|
+
if here.is_file():
|
|
38
|
+
return str(here)
|
|
39
|
+
# Paths are relative to the repository root, so fall back to beside this module.
|
|
40
|
+
beside = CATALOG.parent / Path(str(entry.get("file_name") or here.name))
|
|
41
|
+
return str(beside) if beside.is_file() else ""
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def clip_for(style: str, language: str = "") -> dict[str, Any] | None:
|
|
45
|
+
"""A clip whose style and language match, or None where the catalogue has nothing to offer.
|
|
46
|
+
|
|
47
|
+
Language matters as much as style: a scenario whose mailbox greets in Hindi cannot be served an
|
|
48
|
+
English recording, and no clip is the right answer there because the session speaks the greeting
|
|
49
|
+
itself in the language the scenario asked for.
|
|
50
|
+
|
|
51
|
+
Deterministic rather than random: the first matching entry wins, so two runs of the same suite
|
|
52
|
+
are heard through the same mailbox and a difference between them is never the audio.
|
|
53
|
+
"""
|
|
54
|
+
wanted = str(style or "").strip().lower()
|
|
55
|
+
if not wanted:
|
|
56
|
+
return None
|
|
57
|
+
spoken = (str(language or "").strip().lower() or "en").split("-")[0]
|
|
58
|
+
for entry in _entries():
|
|
59
|
+
if str(entry.get("style") or "").strip().lower() != wanted:
|
|
60
|
+
continue
|
|
61
|
+
if str(entry.get("language") or "en").strip().lower().split("-")[0] != spoken:
|
|
62
|
+
continue
|
|
63
|
+
source = _resolved(entry)
|
|
64
|
+
if not source:
|
|
65
|
+
continue
|
|
66
|
+
return {
|
|
67
|
+
"source": source,
|
|
68
|
+
# A clip that ends with its own tone must not be given a second one.
|
|
69
|
+
"has_tone": bool(entry.get("has_tone")),
|
|
70
|
+
"id": str(entry.get("id") or ""),
|
|
71
|
+
# The greeting must reach the transcript, or the call reads as the agent talking to nobody.
|
|
72
|
+
"transcript": str(entry.get("transcript") or "").strip(),
|
|
73
|
+
}
|
|
74
|
+
return None
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
"""Generated worlds: a real data store behind an agent's tools.
|
|
2
|
+
|
|
3
|
+
The pieces here are the parts that must be exact, so that what gets generated per agent stays
|
|
4
|
+
small: the runtime a world executes on, the snapshot every scenario restores from, and the probe
|
|
5
|
+
suite that decides whether a world is usable at all.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from .kinds import WorldKind, register_kind, supported as supported_kinds
|
|
9
|
+
from .probe import EDGE, HAPPY, SEQUENCE, ProbeReport, ProbeResult, dirty_state, probe
|
|
10
|
+
from .runtime import Call, Db, GeneratedWorld, ToolError, WorldSpec
|
|
11
|
+
from .snapshot import apply_overlay, read_manifest, restore, save
|
|
12
|
+
|
|
13
|
+
__all__ = [
|
|
14
|
+
"Call",
|
|
15
|
+
"Db",
|
|
16
|
+
"EDGE",
|
|
17
|
+
"GeneratedWorld",
|
|
18
|
+
"HAPPY",
|
|
19
|
+
"ProbeReport",
|
|
20
|
+
"ProbeResult",
|
|
21
|
+
"WorldKind",
|
|
22
|
+
"dirty_state",
|
|
23
|
+
"register_kind",
|
|
24
|
+
"supported_kinds",
|
|
25
|
+
"SEQUENCE",
|
|
26
|
+
"ToolError",
|
|
27
|
+
"WorldSpec",
|
|
28
|
+
"apply_overlay",
|
|
29
|
+
"probe",
|
|
30
|
+
"read_manifest",
|
|
31
|
+
"restore",
|
|
32
|
+
"save",
|
|
33
|
+
]
|
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
"""What the hosted world handle raises when scenario code asks it for something it will not do.
|
|
2
|
+
|
|
3
|
+
Every one of these is scenario code at fault, never the world's contents and never the agent
|
|
4
|
+
under test: a `KeyError` from a missing table reads as a finding about data, and a bare
|
|
5
|
+
`StoreError` from the postgres driver reads as an infrastructure fault. Neither is right for
|
|
6
|
+
"you called `change` without saying which column `key` names," so the handle has its own
|
|
7
|
+
vocabulary, and folds it under one base class for whoever has to route "scenario code misused
|
|
8
|
+
the handle" to one outcome without naming all six.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class WorldError(RuntimeError):
|
|
15
|
+
"""Scenario code asked the world handle for something it will not do."""
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class WorldReadOnly(WorldError):
|
|
19
|
+
"""`put`, `change`, `drop` or `call` reached the handle `ready()` or `check()` were given.
|
|
20
|
+
|
|
21
|
+
Those two only ever observe a run. A check that could write would be able to change the very
|
|
22
|
+
thing it is grading, and nothing downstream could tell the difference between a check that
|
|
23
|
+
found a problem and one that quietly fixed it.
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class WorldReservedName(WorldError):
|
|
28
|
+
"""Scenario code named the harness's own conformance canary.
|
|
29
|
+
|
|
30
|
+
That table exists to prove worlds are really isolated from each other, not to hold scenario
|
|
31
|
+
data, and it never appears in `state()` either.
|
|
32
|
+
"""
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class WorldQueryRejected(WorldError):
|
|
36
|
+
"""`query()` was handed something that is not one plain read.
|
|
37
|
+
|
|
38
|
+
The database's own read-only transaction is what actually stops a write; this is the
|
|
39
|
+
friendlier rejection in front of it, so a statement that was never going to be allowed fails
|
|
40
|
+
on a message naming the reason rather than a lock error three layers down.
|
|
41
|
+
"""
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
class WorldStateTooLarge(WorldError):
|
|
45
|
+
"""`state()` reached a table whose row count, measured when the baseline was frozen, passed
|
|
46
|
+
the cap.
|
|
47
|
+
|
|
48
|
+
Measured once, at freeze time, by the provisioner — never recomputed here, so which tables
|
|
49
|
+
raise is fixed before a scenario ever runs and nothing a call does during one can move it.
|
|
50
|
+
"""
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
class WorldUnavailable(WorldError):
|
|
54
|
+
"""The handle cannot do this, given how the world in front of it is built — not what it holds.
|
|
55
|
+
|
|
56
|
+
A postgres world whose `public` schema has no tables, and `call()` — which raises
|
|
57
|
+
unconditionally until the `http_tool` shim's wire format is pinned somewhere in the contracts
|
|
58
|
+
— are both this: nothing went wrong, the capability was never there.
|
|
59
|
+
"""
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
class WorldUsageError(WorldError):
|
|
63
|
+
"""A `put`, `change` or `drop` cannot be carried out as asked.
|
|
64
|
+
|
|
65
|
+
Inserting into something that is not a table, or changing or dropping a record without
|
|
66
|
+
saying which column `key` names — hosted worlds cannot invent tables or guess a column, so
|
|
67
|
+
both are reported here rather than attempted.
|
|
68
|
+
"""
|
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
"""What a world is expected to look like afterwards, and whether it does.
|
|
2
|
+
|
|
3
|
+
Written once and used twice. The build stage declares a sequence and asserts the state it leaves
|
|
4
|
+
behind; a scenario declares the state a conversation should leave behind. Those are the same
|
|
5
|
+
question asked at two different scales, and if each had its own implementation they would drift
|
|
6
|
+
until a check that passes the gate fails the run for reasons that have nothing to do with the
|
|
7
|
+
agent.
|
|
8
|
+
|
|
9
|
+
The shape is ``{"table.count": 3, "table.column": "value"}``: how many records there are, and
|
|
10
|
+
whether a particular value is among them.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
from typing import Any, Mapping
|
|
16
|
+
|
|
17
|
+
COUNT = "count"
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def check_state(
|
|
21
|
+
state: Mapping[str, list[dict[str, Any]]], expected: Mapping[str, Any]
|
|
22
|
+
) -> list[str]:
|
|
23
|
+
"""Every expectation that does not hold, said in terms of what was found instead."""
|
|
24
|
+
failures: list[str] = []
|
|
25
|
+
for path, want in (expected or {}).items():
|
|
26
|
+
table, _, column = str(path).partition(".")
|
|
27
|
+
if table not in state:
|
|
28
|
+
failures.append(
|
|
29
|
+
f"{path}: no {table} in this world; it has "
|
|
30
|
+
f"{', '.join(sorted(state)) or 'nothing'}"
|
|
31
|
+
)
|
|
32
|
+
continue
|
|
33
|
+
rows = state[table]
|
|
34
|
+
if column in ("", COUNT):
|
|
35
|
+
if len(rows) != want:
|
|
36
|
+
failures.append(f"{path}: {len(rows)} rows, expected {want}")
|
|
37
|
+
continue
|
|
38
|
+
if rows and column not in rows[0]:
|
|
39
|
+
failures.append(
|
|
40
|
+
f"{path}: {table} has no {column}; its columns are "
|
|
41
|
+
f"{', '.join(sorted(rows[0]))}"
|
|
42
|
+
)
|
|
43
|
+
continue
|
|
44
|
+
present = {str(row.get(column)) for row in rows}
|
|
45
|
+
# A list means every one of these has to be somewhere, which is how an expectation about
|
|
46
|
+
# a basket of several items is naturally written. Compared as a single value it could
|
|
47
|
+
# never hold, and an expectation that cannot hold grades nothing while appearing to.
|
|
48
|
+
wanted = list(want) if isinstance(want, (list, tuple)) else [want]
|
|
49
|
+
absent = [value for value in wanted if str(value) not in present]
|
|
50
|
+
if absent:
|
|
51
|
+
found = ", ".join(sorted(present)[:6]) or "nothing"
|
|
52
|
+
failures.append(
|
|
53
|
+
f"{path}: no row has {column}="
|
|
54
|
+
+ " or ".join(repr(value) for value in absent)
|
|
55
|
+
+ f"; found {found}"
|
|
56
|
+
)
|
|
57
|
+
return failures
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def unresolvable(
|
|
61
|
+
state: Mapping[str, list[dict[str, Any]]], expected: Mapping[str, Any]
|
|
62
|
+
) -> list[str]:
|
|
63
|
+
"""Expectations that name a table or column the world does not have.
|
|
64
|
+
|
|
65
|
+
Separate from whether they hold, because they are a different kind of wrong. An expectation
|
|
66
|
+
that fails is a finding about the agent; one that names a table nobody built is a finding
|
|
67
|
+
about the expectation, and letting it through means grading a run against a typo.
|
|
68
|
+
"""
|
|
69
|
+
problems: list[str] = []
|
|
70
|
+
for path in expected or {}:
|
|
71
|
+
table, _, column = str(path).partition(".")
|
|
72
|
+
if table not in state:
|
|
73
|
+
# Indexing a particular row is the most common way to write an expectation this
|
|
74
|
+
# cannot carry, and saying only "no such table" sends the reader looking for a
|
|
75
|
+
# spelling mistake instead of at the shape.
|
|
76
|
+
indexed = "[" in table
|
|
77
|
+
problems.append(
|
|
78
|
+
f"{path}: no table called {table!r}"
|
|
79
|
+
+ (
|
|
80
|
+
". Expectations are about the whole table, not one row: use "
|
|
81
|
+
"'table.count' for how many, or 'table.column' for a value that has to "
|
|
82
|
+
"appear in some row."
|
|
83
|
+
if indexed
|
|
84
|
+
else ""
|
|
85
|
+
)
|
|
86
|
+
)
|
|
87
|
+
elif (
|
|
88
|
+
column not in ("", COUNT) and state[table] and column not in state[table][0]
|
|
89
|
+
):
|
|
90
|
+
problems.append(f"{path}: {table} has no column {column!r}")
|
|
91
|
+
return problems
|