agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,457 @@
|
|
|
1
|
+
"""A harness backend that runs stages on Gemini through Vertex AI, on Google's ADK.
|
|
2
|
+
|
|
3
|
+
The Agent Development Kit is Google's counterpart to the Claude Agent SDK: it owns the agentic
|
|
4
|
+
loop, executes tools, and holds session history, the way this harness expects a backend to. We
|
|
5
|
+
adapt at the same seam as the Claude backend and nothing more: a ``ToolSpec`` becomes an ADK
|
|
6
|
+
tool through ADK's own extension point (``BaseTool`` with an explicit declaration), and ADK's
|
|
7
|
+
event stream is translated into the neutral reply vocabulary. The loop itself is not ours.
|
|
8
|
+
|
|
9
|
+
The Gemini 3.x models this exists for are served from the ``global`` endpoint only, which is
|
|
10
|
+
why ``ALK_VERTEX_LOCATION`` defaults to ``global`` rather than to a region. Regional Vertex
|
|
11
|
+
deployments of older models can point it elsewhere.
|
|
12
|
+
|
|
13
|
+
Read, Glob and Grep come from files.py when a stage grants them. AskUserQuestion is not
|
|
14
|
+
implemented here yet: unattended runs never use it, and an attended run on this backend simply
|
|
15
|
+
proceeds without the option, which is said out loud in the session rather than hidden.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import json
|
|
21
|
+
import logging
|
|
22
|
+
import os
|
|
23
|
+
import uuid
|
|
24
|
+
from datetime import date
|
|
25
|
+
from typing import Any, AsyncIterator
|
|
26
|
+
|
|
27
|
+
from .base import (
|
|
28
|
+
FILE_TOOLS,
|
|
29
|
+
Call,
|
|
30
|
+
ModelReply,
|
|
31
|
+
Say,
|
|
32
|
+
SessionOpened,
|
|
33
|
+
SessionSpec,
|
|
34
|
+
StageDone,
|
|
35
|
+
ToolReturned,
|
|
36
|
+
ToolSpec,
|
|
37
|
+
qualified,
|
|
38
|
+
)
|
|
39
|
+
from .files import file_tools
|
|
40
|
+
|
|
41
|
+
DEFAULT_MODEL = "gemini-3.7-flash"
|
|
42
|
+
|
|
43
|
+
_TERMINAL_SAVE_TOOLS = frozenset(
|
|
44
|
+
{
|
|
45
|
+
"mcp__world__save_world",
|
|
46
|
+
"mcp__provision__save_environment",
|
|
47
|
+
"mcp__scenarios__save_scenarios",
|
|
48
|
+
# Source-data review is another persisted authoring boundary. Its handler only
|
|
49
|
+
# succeeds after every scenario was reviewed and at least one executable invariant
|
|
50
|
+
# was declared. Letting ADK take another turn after that success can burn the entire
|
|
51
|
+
# call budget and turn a completed review into a spurious validation failure.
|
|
52
|
+
"mcp__source_data__finish_review",
|
|
53
|
+
}
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
# Vertex list pricing per 1M tokens: (input, output, the day this pair was last checked against
|
|
57
|
+
# the platform's litellm model table). An unknown or stale model reports no cost rather than a
|
|
58
|
+
# wrong one, and shows up in `unpriced_turns`.
|
|
59
|
+
PRICES_PER_MILLION = {
|
|
60
|
+
"gemini-3.8-flash": (0.75, 3.75, "2026-12-31"),
|
|
61
|
+
"gemini-3.7-flash": (0.75, 3.75, "2026-12-31"),
|
|
62
|
+
"gemini-3.6-flash": (0.75, 3.75, "2026-12-31"),
|
|
63
|
+
"gemini-3.5-transcribe-preview": (2.5, 12, "2026-12-31"),
|
|
64
|
+
"gemini-3.5-transcribe-live-preview": (3.5, 21, "2026-12-31"),
|
|
65
|
+
"gemini-3.5-flash-lite": (0.3, 2.5, "2026-12-31"),
|
|
66
|
+
"gemini-3.5-flash": (1.5, 9, "2026-12-31"),
|
|
67
|
+
"gemini-3.1-pro-preview-customtools": (2, 12, "2026-12-31"),
|
|
68
|
+
"gemini-3.1-pro-preview": (2, 12, "2026-12-31"),
|
|
69
|
+
"gemini-3.1-flash-lite-preview": (0.25, 1.5, "2026-12-31"),
|
|
70
|
+
"gemini-3.1-flash-lite-image": (0.25, 1.5, "2026-12-31"),
|
|
71
|
+
"gemini-3.1-flash-lite": (0.25, 1.5, "2026-12-31"),
|
|
72
|
+
"gemini-3.1-flash-image-preview": (0.5, 3, "2026-12-31"),
|
|
73
|
+
"gemini-3.1-flash-image": (0.5, 3, "2026-12-31"),
|
|
74
|
+
"gemini-3-pro-preview": (2, 12, "2026-12-31"),
|
|
75
|
+
"gemini-3-pro-image-preview": (2, 12, "2026-12-31"),
|
|
76
|
+
"gemini-3-pro-image": (2, 12, "2026-12-31"),
|
|
77
|
+
"gemini-3-flash-preview": (0.5, 3, "2026-12-31"),
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
_PYTHON_TYPES = {
|
|
81
|
+
str: "string",
|
|
82
|
+
int: "integer",
|
|
83
|
+
float: "number",
|
|
84
|
+
bool: "boolean",
|
|
85
|
+
list: "array",
|
|
86
|
+
dict: "object",
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
# The keys Vertex's Schema type accepts. JSON Schema carries more; anything else is dropped
|
|
90
|
+
# rather than passed through, because an unknown key fails declaration validation and takes
|
|
91
|
+
# the whole stage down before its first turn.
|
|
92
|
+
_SCHEMA_KEYS = (
|
|
93
|
+
"description",
|
|
94
|
+
"enum",
|
|
95
|
+
"format",
|
|
96
|
+
"items",
|
|
97
|
+
"maximum",
|
|
98
|
+
"maxItems",
|
|
99
|
+
"minimum",
|
|
100
|
+
"minItems",
|
|
101
|
+
"nullable",
|
|
102
|
+
"pattern",
|
|
103
|
+
"properties",
|
|
104
|
+
"required",
|
|
105
|
+
"type",
|
|
106
|
+
"anyOf",
|
|
107
|
+
)
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def _gemini_schema(schema: Any) -> dict[str, Any]:
|
|
111
|
+
"""A JSON Schema fragment as Vertex's Schema dialect.
|
|
112
|
+
|
|
113
|
+
The real difference is nullability: JSON Schema says ``"type": ["string", "null"]``,
|
|
114
|
+
Vertex says ``"type": "string", "nullable": true``. Everything Vertex does not know is
|
|
115
|
+
dropped, recursively, so a tool schema written for the loosest backend still declares.
|
|
116
|
+
"""
|
|
117
|
+
if not isinstance(schema, dict):
|
|
118
|
+
return {"type": "string"}
|
|
119
|
+
cleaned: dict[str, Any] = {}
|
|
120
|
+
for key, value in schema.items():
|
|
121
|
+
if key not in _SCHEMA_KEYS:
|
|
122
|
+
continue
|
|
123
|
+
if key == "type" and isinstance(value, list):
|
|
124
|
+
bare = [entry for entry in value if entry != "null"]
|
|
125
|
+
cleaned["type"] = bare[0] if bare else "string"
|
|
126
|
+
if "null" in value:
|
|
127
|
+
cleaned["nullable"] = True
|
|
128
|
+
elif key == "properties" and isinstance(value, dict):
|
|
129
|
+
cleaned["properties"] = {
|
|
130
|
+
name: _gemini_schema(inner) for name, inner in value.items()
|
|
131
|
+
}
|
|
132
|
+
elif key == "items":
|
|
133
|
+
cleaned["items"] = _gemini_schema(value)
|
|
134
|
+
elif key == "anyOf" and isinstance(value, list):
|
|
135
|
+
cleaned["anyOf"] = [_gemini_schema(inner) for inner in value]
|
|
136
|
+
elif key == "enum" and isinstance(value, list):
|
|
137
|
+
# Vertex enums are strings; None inside one is JSON Schema's way of saying
|
|
138
|
+
# nullable, and everything else is stringified the way the model will echo it.
|
|
139
|
+
cleaned["enum"] = [str(entry) for entry in value if entry is not None]
|
|
140
|
+
if None in value:
|
|
141
|
+
cleaned["nullable"] = True
|
|
142
|
+
else:
|
|
143
|
+
cleaned[key] = value
|
|
144
|
+
# Vertex refuses an array that does not say what it holds. JSON Schema treats items as
|
|
145
|
+
# optional, and most such arrays here carry row-shaped dicts, so an open object is the
|
|
146
|
+
# faithful default.
|
|
147
|
+
if cleaned.get("type") == "array" and "items" not in cleaned:
|
|
148
|
+
cleaned["items"] = {"type": "object"}
|
|
149
|
+
return cleaned
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def _json_schema(schema: Any) -> dict[str, Any]:
|
|
153
|
+
"""The tool's schema as Vertex-safe schema, whichever shorthand it was declared in."""
|
|
154
|
+
if isinstance(schema, dict) and (
|
|
155
|
+
"properties" in schema or schema.get("type") == "object"
|
|
156
|
+
):
|
|
157
|
+
return _gemini_schema(schema)
|
|
158
|
+
if isinstance(schema, dict):
|
|
159
|
+
return {
|
|
160
|
+
"type": "object",
|
|
161
|
+
"properties": {
|
|
162
|
+
name: (
|
|
163
|
+
{"type": "array", "items": {"type": "object"}}
|
|
164
|
+
if kind is list
|
|
165
|
+
else {"type": _PYTHON_TYPES.get(kind, "string")}
|
|
166
|
+
)
|
|
167
|
+
for name, kind in schema.items()
|
|
168
|
+
},
|
|
169
|
+
"required": list(schema),
|
|
170
|
+
}
|
|
171
|
+
return {"type": "object", "properties": {}}
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def _project() -> str:
|
|
175
|
+
named = os.environ.get("GOOGLE_CLOUD_PROJECT")
|
|
176
|
+
if named:
|
|
177
|
+
return named
|
|
178
|
+
credentials = os.environ.get("GOOGLE_APPLICATION_CREDENTIALS")
|
|
179
|
+
if credentials:
|
|
180
|
+
try:
|
|
181
|
+
with open(credentials, encoding="utf-8") as handle:
|
|
182
|
+
found = json.load(handle).get("project_id", "")
|
|
183
|
+
if found:
|
|
184
|
+
return found
|
|
185
|
+
except (OSError, ValueError):
|
|
186
|
+
pass
|
|
187
|
+
# Application Default Credentials already carry a project on a machine that has run
|
|
188
|
+
# ``gcloud auth application-default login`` or that runs on Google infrastructure. Asking
|
|
189
|
+
# for it again as an environment variable is a setting the operator does not need to know.
|
|
190
|
+
try:
|
|
191
|
+
import google.auth
|
|
192
|
+
|
|
193
|
+
_, discovered = google.auth.default()
|
|
194
|
+
if discovered:
|
|
195
|
+
return str(discovered)
|
|
196
|
+
except Exception: # noqa: BLE001 - fall through to the explicit instruction below
|
|
197
|
+
pass
|
|
198
|
+
raise RuntimeError(
|
|
199
|
+
"no GCP project named; run 'gcloud auth application-default login', or set "
|
|
200
|
+
"GOOGLE_CLOUD_PROJECT, or point GOOGLE_APPLICATION_CREDENTIALS at a service-account file"
|
|
201
|
+
)
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def _location() -> str:
|
|
205
|
+
return os.environ.get("ALK_VERTEX_LOCATION", "global").strip() or "global"
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
def _flattened(result: Any) -> str:
|
|
209
|
+
content = result.get("content") if isinstance(result, dict) else None
|
|
210
|
+
if isinstance(content, list):
|
|
211
|
+
return "\n".join(
|
|
212
|
+
part.get("text", "") for part in content if isinstance(part, dict)
|
|
213
|
+
)
|
|
214
|
+
return content if isinstance(content, str) else str(result)
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
def _successful_terminal_save(name: str, response: Any) -> bool:
|
|
218
|
+
"""Whether a tool response proves this authoring stage has persisted its final output.
|
|
219
|
+
|
|
220
|
+
Save tools deliberately reject incomplete work with ``is_error`` so the model can repair and
|
|
221
|
+
retry. Once one succeeds, another model turn can only rewrite already-valid output or burn the
|
|
222
|
+
stage budget; the persisted artifact is the stage's actual completion boundary.
|
|
223
|
+
"""
|
|
224
|
+
return bool(
|
|
225
|
+
name in _TERMINAL_SAVE_TOOLS
|
|
226
|
+
and isinstance(response, dict)
|
|
227
|
+
and not response.get("is_error")
|
|
228
|
+
)
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
def _spec_tool(name: str, spec: ToolSpec) -> Any:
|
|
232
|
+
"""A ToolSpec as an ADK tool, through ADK's own extension point.
|
|
233
|
+
|
|
234
|
+
``BaseTool`` with an explicit ``_get_declaration`` is how ADK says a tool whose contract
|
|
235
|
+
is defined elsewhere should be wrapped; the handler runs unchanged and ADK owns calling
|
|
236
|
+
it, retrying the turn, and feeding the result back.
|
|
237
|
+
"""
|
|
238
|
+
from google.adk.tools import BaseTool
|
|
239
|
+
from google.genai import types
|
|
240
|
+
|
|
241
|
+
class SpecTool(BaseTool):
|
|
242
|
+
def __init__(self) -> None:
|
|
243
|
+
super().__init__(name=name, description=spec.description)
|
|
244
|
+
|
|
245
|
+
def _get_declaration(self) -> Any:
|
|
246
|
+
return types.FunctionDeclaration(
|
|
247
|
+
name=name,
|
|
248
|
+
description=spec.description,
|
|
249
|
+
parameters=_json_schema(spec.input_schema),
|
|
250
|
+
)
|
|
251
|
+
|
|
252
|
+
async def run_async(self, *, args: dict[str, Any], tool_context: Any) -> Any:
|
|
253
|
+
return await spec.handler(args)
|
|
254
|
+
|
|
255
|
+
return SpecTool()
|
|
256
|
+
|
|
257
|
+
|
|
258
|
+
class VertexGeminiSession:
|
|
259
|
+
"""One ADK-run conversation with Gemini; ADK holds the history across turns."""
|
|
260
|
+
|
|
261
|
+
def __init__(self, spec: SessionSpec, model: str) -> None:
|
|
262
|
+
self._spec = spec
|
|
263
|
+
self._model = model
|
|
264
|
+
self._runner: Any = None
|
|
265
|
+
self._pending: str | None = None
|
|
266
|
+
self.session_id = f"gemini-{uuid.uuid4().hex[:12]}"
|
|
267
|
+
|
|
268
|
+
def _tools(self) -> list[Any]:
|
|
269
|
+
# ASK_TOOL is deliberately absent: unattended runs never call it, and declaring a tool
|
|
270
|
+
# this backend cannot answer would cost the model a turn finding that out.
|
|
271
|
+
offered: list[Any] = []
|
|
272
|
+
wanted = {name for name in self._spec.builtins if name in FILE_TOOLS}
|
|
273
|
+
offered.extend(
|
|
274
|
+
_spec_tool(spec.name, spec)
|
|
275
|
+
for spec in file_tools(self._spec.cwd)
|
|
276
|
+
if spec.name in wanted
|
|
277
|
+
)
|
|
278
|
+
for server_name, server in self._spec.servers.items():
|
|
279
|
+
offered.extend(
|
|
280
|
+
_spec_tool(qualified(server_name, spec.name), spec)
|
|
281
|
+
for spec in server.tools
|
|
282
|
+
)
|
|
283
|
+
return offered
|
|
284
|
+
|
|
285
|
+
async def start(self) -> None:
|
|
286
|
+
from google.adk.agents import LlmAgent
|
|
287
|
+
from google.adk.runners import Runner
|
|
288
|
+
from google.adk.sessions import InMemorySessionService
|
|
289
|
+
from google.genai import types
|
|
290
|
+
|
|
291
|
+
# ADK builds its Vertex client from the environment, the same way the Claude backend
|
|
292
|
+
# passes provider env through its options.
|
|
293
|
+
os.environ["GOOGLE_GENAI_USE_VERTEXAI"] = "TRUE"
|
|
294
|
+
os.environ["GOOGLE_CLOUD_PROJECT"] = _project()
|
|
295
|
+
os.environ["GOOGLE_CLOUD_LOCATION"] = _location()
|
|
296
|
+
# static_instruction, not instruction: the skills are full of literal JSON braces,
|
|
297
|
+
# and ADK templates {placeholders} in `instruction` from session state. Static
|
|
298
|
+
# content is sent verbatim and is what ADK context-caches.
|
|
299
|
+
agent = LlmAgent(
|
|
300
|
+
name=self.session_id.replace("-", "_"),
|
|
301
|
+
model=self._model,
|
|
302
|
+
static_instruction=types.Content(
|
|
303
|
+
role="user", parts=[types.Part(text=self._spec.system_prompt)]
|
|
304
|
+
),
|
|
305
|
+
tools=self._tools(),
|
|
306
|
+
)
|
|
307
|
+
sessions = InMemorySessionService()
|
|
308
|
+
await sessions.create_session(
|
|
309
|
+
app_name="alk-harness", user_id="stage", session_id=self.session_id
|
|
310
|
+
)
|
|
311
|
+
self._runner = Runner(
|
|
312
|
+
agent=agent, app_name="alk-harness", session_service=sessions
|
|
313
|
+
)
|
|
314
|
+
|
|
315
|
+
async def stop(self) -> None:
|
|
316
|
+
if self._runner is not None:
|
|
317
|
+
await self._runner.close()
|
|
318
|
+
self._runner = None
|
|
319
|
+
|
|
320
|
+
async def send(self, message: str) -> None:
|
|
321
|
+
if self._runner is None:
|
|
322
|
+
raise RuntimeError("session is not open")
|
|
323
|
+
self._pending = message
|
|
324
|
+
|
|
325
|
+
async def replies(self) -> AsyncIterator[Any]:
|
|
326
|
+
from google.adk.agents.run_config import RunConfig
|
|
327
|
+
from google.genai import types
|
|
328
|
+
|
|
329
|
+
if self._runner is None or self._pending is None:
|
|
330
|
+
raise RuntimeError("nothing to reply to; send a message first")
|
|
331
|
+
yield SessionOpened(session_id=self.session_id)
|
|
332
|
+
message = types.Content(role="user", parts=[types.Part(text=self._pending)])
|
|
333
|
+
self._pending = None
|
|
334
|
+
turns = 0
|
|
335
|
+
tokens_in = 0
|
|
336
|
+
tokens_out = 0
|
|
337
|
+
tokens_cached = 0
|
|
338
|
+
settled = False
|
|
339
|
+
terminal_save_succeeded = False
|
|
340
|
+
try:
|
|
341
|
+
async for event in self._runner.run_async(
|
|
342
|
+
user_id="stage",
|
|
343
|
+
session_id=self.session_id,
|
|
344
|
+
new_message=message,
|
|
345
|
+
run_config=RunConfig(max_llm_calls=max(self._spec.max_turns, 1)),
|
|
346
|
+
):
|
|
347
|
+
usage = getattr(event, "usage_metadata", None)
|
|
348
|
+
if usage is not None:
|
|
349
|
+
tokens_in += usage.prompt_token_count or 0
|
|
350
|
+
tokens_out += usage.candidates_token_count or 0
|
|
351
|
+
tokens_cached += getattr(usage, "cached_content_token_count", 0) or 0
|
|
352
|
+
parts: list[Any] = []
|
|
353
|
+
returned: list[ToolReturned] = []
|
|
354
|
+
for part in (event.content.parts if event.content else []) or []:
|
|
355
|
+
if getattr(part, "text", None):
|
|
356
|
+
parts.append(Say(text=part.text))
|
|
357
|
+
if getattr(part, "function_call", None):
|
|
358
|
+
parts.append(
|
|
359
|
+
Call(
|
|
360
|
+
id=getattr(part.function_call, "id", None)
|
|
361
|
+
or f"call-{uuid.uuid4().hex[:8]}",
|
|
362
|
+
name=part.function_call.name or "",
|
|
363
|
+
arguments=dict(part.function_call.args or {}),
|
|
364
|
+
)
|
|
365
|
+
)
|
|
366
|
+
if getattr(part, "function_response", None):
|
|
367
|
+
response = part.function_response.response
|
|
368
|
+
response_name = part.function_response.name or ""
|
|
369
|
+
terminal_save_succeeded = terminal_save_succeeded or (
|
|
370
|
+
_successful_terminal_save(response_name, response)
|
|
371
|
+
)
|
|
372
|
+
returned.append(
|
|
373
|
+
ToolReturned(
|
|
374
|
+
id=getattr(part.function_response, "id", None) or "",
|
|
375
|
+
text=_flattened(response),
|
|
376
|
+
is_error=bool(
|
|
377
|
+
isinstance(response, dict)
|
|
378
|
+
and response.get("is_error")
|
|
379
|
+
),
|
|
380
|
+
)
|
|
381
|
+
)
|
|
382
|
+
if parts:
|
|
383
|
+
turns += 1
|
|
384
|
+
yield ModelReply(parts=parts, model=self._model)
|
|
385
|
+
for outcome in returned:
|
|
386
|
+
yield outcome
|
|
387
|
+
if terminal_save_succeeded:
|
|
388
|
+
settled = True
|
|
389
|
+
break
|
|
390
|
+
if event.is_final_response():
|
|
391
|
+
settled = True
|
|
392
|
+
except Exception as exc:
|
|
393
|
+
yield StageDone(
|
|
394
|
+
outcome="failed",
|
|
395
|
+
turns=turns,
|
|
396
|
+
cost_usd=self._cost(tokens_in, tokens_out),
|
|
397
|
+
tokens_in=tokens_in,
|
|
398
|
+
tokens_out=tokens_out,
|
|
399
|
+
tokens_cached=tokens_cached,
|
|
400
|
+
session_id=self.session_id,
|
|
401
|
+
models={self._model},
|
|
402
|
+
is_error=True,
|
|
403
|
+
api_error_status=getattr(exc, "code", None),
|
|
404
|
+
errors=[str(exc)[:400]],
|
|
405
|
+
)
|
|
406
|
+
return
|
|
407
|
+
# A stream that ends without a final response ran out of its call budget, which is
|
|
408
|
+
# not the same as the model having finished. Reported as success it reads as a stage
|
|
409
|
+
# that did its work, and a half-written suite comes back green.
|
|
410
|
+
yield StageDone(
|
|
411
|
+
outcome="success" if settled else "max_turns",
|
|
412
|
+
is_error=not settled,
|
|
413
|
+
turns=turns,
|
|
414
|
+
cost_usd=self._cost(tokens_in, tokens_out),
|
|
415
|
+
tokens_in=tokens_in,
|
|
416
|
+
tokens_out=tokens_out,
|
|
417
|
+
tokens_cached=tokens_cached,
|
|
418
|
+
session_id=self.session_id,
|
|
419
|
+
models={self._model},
|
|
420
|
+
errors=(
|
|
421
|
+
[]
|
|
422
|
+
if settled
|
|
423
|
+
else [
|
|
424
|
+
f"the stage spent its whole budget of {self._spec.max_turns} calls"
|
|
425
|
+
]
|
|
426
|
+
),
|
|
427
|
+
)
|
|
428
|
+
|
|
429
|
+
def _cost(self, tokens_in: int, tokens_out: int) -> float | None:
|
|
430
|
+
return priced(self._model, tokens_in, tokens_out)
|
|
431
|
+
|
|
432
|
+
|
|
433
|
+
logger = logging.getLogger(__name__)
|
|
434
|
+
|
|
435
|
+
|
|
436
|
+
def priced(model: str, tokens_in: int, tokens_out: int) -> float | None:
|
|
437
|
+
"""What these tokens cost, or None where no price can be stood behind."""
|
|
438
|
+
prices = PRICES_PER_MILLION.get(model)
|
|
439
|
+
if prices is None:
|
|
440
|
+
return None
|
|
441
|
+
if len(prices) > 2 and date.today().isoformat() > str(prices[2]):
|
|
442
|
+
logger.warning(
|
|
443
|
+
"no current price for %s: the table's figures expired on %s", model, prices[2]
|
|
444
|
+
)
|
|
445
|
+
return None
|
|
446
|
+
return (tokens_in * prices[0] + tokens_out * prices[1]) / 1_000_000
|
|
447
|
+
|
|
448
|
+
|
|
449
|
+
class VertexGeminiBackend:
|
|
450
|
+
name = "vertex-gemini"
|
|
451
|
+
default_model = DEFAULT_MODEL
|
|
452
|
+
|
|
453
|
+
def can_drive(self, model: str) -> bool:
|
|
454
|
+
return (model or "").lower().startswith("gemini")
|
|
455
|
+
|
|
456
|
+
def create(self, spec: SessionSpec) -> VertexGeminiSession:
|
|
457
|
+
return VertexGeminiSession(spec, model=spec.model or self.default_model)
|
|
@@ -0,0 +1,95 @@
|
|
|
1
|
+
"""Choose the caller-side ambient noise a scenario should be heard through.
|
|
2
|
+
|
|
3
|
+
A scenario that sets ``background_noise`` wants the agent to handle a caller phoning from somewhere
|
|
4
|
+
real: a car, a street, an office. The clip is chosen here and handed to the voice engine, which
|
|
5
|
+
mixes it under the simulated caller's audio.
|
|
6
|
+
|
|
7
|
+
Two sources, in order. A run may point ``ALK_BACKGROUND_NOISE_CATALOG`` at a JSON file of clips
|
|
8
|
+
(each with an ``environment`` tag and a ``url`` or ``path``); the catalog stays a local file so its
|
|
9
|
+
asset locations are never committed here. When no catalog matches, a LiveKit builtin clip is used,
|
|
10
|
+
which needs no external asset and always works.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import json
|
|
16
|
+
import os
|
|
17
|
+
from pathlib import Path
|
|
18
|
+
|
|
19
|
+
# LiveKit ships these; they are the reliable default when no custom catalog is configured.
|
|
20
|
+
_BUILTIN_BY_ENVIRONMENT: dict[str, str] = {
|
|
21
|
+
"street": "CITY_AMBIENCE",
|
|
22
|
+
"transit": "CITY_AMBIENCE",
|
|
23
|
+
"vehicle": "CITY_AMBIENCE",
|
|
24
|
+
"outdoors": "FOREST_AMBIENCE",
|
|
25
|
+
"retail": "CROWDED_ROOM",
|
|
26
|
+
"office": "OFFICE_AMBIENCE",
|
|
27
|
+
"home": "OFFICE_AMBIENCE",
|
|
28
|
+
}
|
|
29
|
+
_DEFAULT_BUILTIN = "OFFICE_AMBIENCE"
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def enabled() -> bool:
|
|
33
|
+
"""Whether any scenario may be heard through background noise on this run.
|
|
34
|
+
|
|
35
|
+
**On unless ``ALK_BACKGROUND_NOISE`` turns it off.** A real caller is somewhere, and an agent
|
|
36
|
+
tested only against studio silence has not been tested against its callers, so noise is what a
|
|
37
|
+
run should fall into rather than something it has to ask for. It was opt-in and every deployment
|
|
38
|
+
forgot: a switch nobody sets is a feature nobody has.
|
|
39
|
+
|
|
40
|
+
The tradeoff is real and is why an opt-out exists. Continuous ambience under the caller competes
|
|
41
|
+
with endpoint detection, and calls carrying it end a little earlier and on fewer turns. Set
|
|
42
|
+
``ALK_BACKGROUND_NOISE=0`` (or ``off``, ``false``, ``no``) for a run that needs a clean line.
|
|
43
|
+
|
|
44
|
+
Permission, not compulsion: a scenario whose own ``background_noise`` says none stays silent
|
|
45
|
+
either way.
|
|
46
|
+
"""
|
|
47
|
+
return os.environ.get("ALK_BACKGROUND_NOISE", "1").strip().lower() not in (
|
|
48
|
+
"0",
|
|
49
|
+
"off",
|
|
50
|
+
"false",
|
|
51
|
+
"no",
|
|
52
|
+
)
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def source_for(environment: str = "", seed: str = "") -> str:
|
|
56
|
+
"""A background-noise source for a scenario.
|
|
57
|
+
|
|
58
|
+
Returns a ``url``/``path`` from the configured catalog when one matches the environment, else the
|
|
59
|
+
name of a LiveKit builtin clip. The choice is deterministic in ``seed`` so the same scenario
|
|
60
|
+
hears the same place across runs.
|
|
61
|
+
"""
|
|
62
|
+
env = (environment or "").strip().lower()
|
|
63
|
+
catalog = os.environ.get("ALK_BACKGROUND_NOISE_CATALOG", "").strip()
|
|
64
|
+
if catalog and Path(catalog).is_file():
|
|
65
|
+
try:
|
|
66
|
+
entries = json.loads(Path(catalog).read_text(encoding="utf-8"))
|
|
67
|
+
except (OSError, ValueError):
|
|
68
|
+
entries = []
|
|
69
|
+
if isinstance(entries, list) and entries:
|
|
70
|
+
pool = [
|
|
71
|
+
entry
|
|
72
|
+
for entry in entries
|
|
73
|
+
if str(entry.get("environment", "")).strip().lower() == env
|
|
74
|
+
] or entries
|
|
75
|
+
chosen = pool[sum(ord(character) for character in (seed or env or "x")) % len(pool)]
|
|
76
|
+
located = str(chosen.get("url") or chosen.get("path") or "").strip()
|
|
77
|
+
if located:
|
|
78
|
+
return located
|
|
79
|
+
return _BUILTIN_BY_ENVIRONMENT.get(env, _DEFAULT_BUILTIN)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def scenario_source(
|
|
83
|
+
background_noise, fixture, seed: str = ""
|
|
84
|
+
) -> str:
|
|
85
|
+
"""The noise source for one scenario, or "" when it should be heard in the clear.
|
|
86
|
+
|
|
87
|
+
The scenario names the place when it cares which one; otherwise the fixture says where the
|
|
88
|
+
caller is, and failing that any noise will do.
|
|
89
|
+
"""
|
|
90
|
+
if not background_noise or not enabled():
|
|
91
|
+
return ""
|
|
92
|
+
environment = background_noise if isinstance(background_noise, str) else ""
|
|
93
|
+
if not environment and isinstance(fixture, dict):
|
|
94
|
+
environment = str(fixture.get("environment") or fixture.get("location") or "")
|
|
95
|
+
return source_for(environment, seed=seed)
|