agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,517 @@
|
|
|
1
|
+
import asyncio
|
|
2
|
+
import os
|
|
3
|
+
import contextvars
|
|
4
|
+
import logging
|
|
5
|
+
import contextlib
|
|
6
|
+
from typing import Optional, Callable
|
|
7
|
+
|
|
8
|
+
from fi.simulate._logging import redacted_exc_info
|
|
9
|
+
from fi.simulate.agent.generic import wrap_agent
|
|
10
|
+
from fi.simulate.agent.wrapper import AgentWrapper, AgentInput, AgentResponse
|
|
11
|
+
from fi.simulate.simulation.models import TestReport
|
|
12
|
+
from fi.simulate.simulation.engines.base import BaseEngine
|
|
13
|
+
from fi.simulate.utils.routes import APIRoutes
|
|
14
|
+
|
|
15
|
+
# Context variable to track the current execution ID for future tool mocking
|
|
16
|
+
current_execution_id = contextvars.ContextVar("current_execution_id", default=None)
|
|
17
|
+
|
|
18
|
+
logger = logging.getLogger(__name__)
|
|
19
|
+
|
|
20
|
+
class CloudEngine(BaseEngine):
|
|
21
|
+
"""
|
|
22
|
+
Execution engine that connects to the Future AGI backend to orchestrate simulations.
|
|
23
|
+
It acts as a bridge between the cloud-hosted simulator and the user's local agent.
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
def __init__(self, api_key: Optional[str] = None, secret_key: Optional[str] = None, api_url: Optional[str] = None, timeout: float = 120.0):
|
|
27
|
+
"""
|
|
28
|
+
Args:
|
|
29
|
+
api_key: API key for authentication
|
|
30
|
+
secret_key: Secret key for authentication
|
|
31
|
+
api_url: Base URL of the backend API
|
|
32
|
+
timeout: Request timeout in seconds (default: 120s for LLM operations)
|
|
33
|
+
"""
|
|
34
|
+
self.api_key = api_key or os.environ.get("FI_API_KEY")
|
|
35
|
+
self.secret_key = secret_key or os.environ.get("FI_SECRET_KEY")
|
|
36
|
+
self.api_url = api_url or os.environ.get("FI_BASE_URL") or "https://api.futureagi.com"
|
|
37
|
+
self.timeout = timeout
|
|
38
|
+
|
|
39
|
+
if not self.api_key or not self.secret_key:
|
|
40
|
+
logger.warning("FI_API_KEY or FI_SECRET_KEY not provided. CloudEngine will not function correctly.")
|
|
41
|
+
|
|
42
|
+
self.api = None
|
|
43
|
+
self.run_test_id = None
|
|
44
|
+
self.test_execution_id = None
|
|
45
|
+
self._using_simulator_attributes = None
|
|
46
|
+
try:
|
|
47
|
+
# Optional dependency: enables baggage propagation so user spans inherit simulator IDs
|
|
48
|
+
from fi_instrumentation import using_simulator_attributes # type: ignore
|
|
49
|
+
self._using_simulator_attributes = using_simulator_attributes
|
|
50
|
+
except Exception:
|
|
51
|
+
self._using_simulator_attributes = None
|
|
52
|
+
|
|
53
|
+
async def run(
|
|
54
|
+
self,
|
|
55
|
+
run_id: Optional[str] = None,
|
|
56
|
+
run_test_name: Optional[str] = None,
|
|
57
|
+
agent_callback: Optional[Callable | AgentWrapper] = None,
|
|
58
|
+
concurrency: int = 5,
|
|
59
|
+
**kwargs
|
|
60
|
+
) -> TestReport:
|
|
61
|
+
"""
|
|
62
|
+
Connects to the cloud run, receives user inputs, calls the agent_callback,
|
|
63
|
+
and sends responses back.
|
|
64
|
+
"""
|
|
65
|
+
if not run_id and not run_test_name:
|
|
66
|
+
raise ValueError("CloudEngine requires either 'run_id' or 'run_test_name'.")
|
|
67
|
+
|
|
68
|
+
if not agent_callback:
|
|
69
|
+
raise ValueError("CloudEngine requires an 'agent_callback' (function or AgentWrapper).")
|
|
70
|
+
|
|
71
|
+
self.api = APIRoutes(self.api_key, self.secret_key, self.api_url, timeout=self.timeout)
|
|
72
|
+
|
|
73
|
+
# If run_test_name is provided, fetch the run_id first
|
|
74
|
+
if run_test_name and not run_id:
|
|
75
|
+
print(f"🔍 Fetching Run Test ID for name: {run_test_name}")
|
|
76
|
+
try:
|
|
77
|
+
name_resp = await self.api.get_run_test_id_by_name(run_test_name)
|
|
78
|
+
result = name_resp.get("result", {})
|
|
79
|
+
# Handle both camelCase and snake_case response formats
|
|
80
|
+
run_id = result.get("run_test_id") or result.get("runTestId")
|
|
81
|
+
if not run_id:
|
|
82
|
+
raise ValueError(f"Failed to get run_test_id for name '{run_test_name}'. Response: {name_resp}")
|
|
83
|
+
print(f"✓ Found Run Test ID: {run_id}")
|
|
84
|
+
except Exception as e:
|
|
85
|
+
logger.error(f"Failed to get run_test_id by name: {e}")
|
|
86
|
+
raise ValueError(f"Failed to get run_test_id for name '{run_test_name}': {e}")
|
|
87
|
+
|
|
88
|
+
wrapper = self._normalize_callback(agent_callback)
|
|
89
|
+
queue = asyncio.Queue()
|
|
90
|
+
|
|
91
|
+
# Store IDs for tracing attributes
|
|
92
|
+
self.run_test_id = run_id
|
|
93
|
+
|
|
94
|
+
print(f"Starting Simulation for Run ID: {run_id}")
|
|
95
|
+
|
|
96
|
+
try:
|
|
97
|
+
# 1. Start the Run (Create TestExecution)
|
|
98
|
+
start_resp = await self.api.start_test_execution(run_test_id=run_id)
|
|
99
|
+
result = start_resp.get("result", {})
|
|
100
|
+
# Handle both camelCase and snake_case response formats
|
|
101
|
+
test_execution_id = result.get("executionId") or result.get("execution_id")
|
|
102
|
+
|
|
103
|
+
if not test_execution_id:
|
|
104
|
+
raise ValueError(f"Failed to start test execution. Response: {start_resp}")
|
|
105
|
+
|
|
106
|
+
print(f"✓ Test Execution Started: {test_execution_id}")
|
|
107
|
+
|
|
108
|
+
# Store test execution ID for tracing
|
|
109
|
+
self.test_execution_id = test_execution_id
|
|
110
|
+
|
|
111
|
+
# 2. Start Producer and Consumers
|
|
112
|
+
producer_task = asyncio.create_task(
|
|
113
|
+
self._producer_loop(run_id, test_execution_id, queue)
|
|
114
|
+
)
|
|
115
|
+
|
|
116
|
+
consumers = [
|
|
117
|
+
asyncio.create_task(self._consumer_loop(queue, wrapper))
|
|
118
|
+
for _ in range(concurrency)
|
|
119
|
+
]
|
|
120
|
+
|
|
121
|
+
# Wait for producer to finish fetching all batches
|
|
122
|
+
await producer_task
|
|
123
|
+
|
|
124
|
+
# Wait for queue to drain (all consumers process remaining items)
|
|
125
|
+
await queue.join()
|
|
126
|
+
|
|
127
|
+
# Cancel consumers
|
|
128
|
+
for c in consumers:
|
|
129
|
+
c.cancel()
|
|
130
|
+
|
|
131
|
+
print("✅ Cloud Simulation Completed.")
|
|
132
|
+
|
|
133
|
+
except Exception as exc:
|
|
134
|
+
logger.error(
|
|
135
|
+
"Cloud simulation failed",
|
|
136
|
+
exc_info=redacted_exc_info(exc),
|
|
137
|
+
extra={"exception_type": type(exc).__name__},
|
|
138
|
+
)
|
|
139
|
+
raise
|
|
140
|
+
finally:
|
|
141
|
+
if self.api:
|
|
142
|
+
await self.api.close()
|
|
143
|
+
|
|
144
|
+
# Return empty report for now as backend handles metrics
|
|
145
|
+
return TestReport(results=[])
|
|
146
|
+
|
|
147
|
+
async def _producer_loop(self, run_test_id: str, test_execution_id: str, queue: asyncio.Queue):
|
|
148
|
+
"""
|
|
149
|
+
Polls the backend for batches of call execution IDs and puts them in the queue.
|
|
150
|
+
"""
|
|
151
|
+
has_more = True
|
|
152
|
+
|
|
153
|
+
while has_more:
|
|
154
|
+
try:
|
|
155
|
+
print("🔄 Fetching batch of scenarios...")
|
|
156
|
+
resp = await self.api.fetch_execution_batch(test_execution_id)
|
|
157
|
+
|
|
158
|
+
result = resp.get("result", {})
|
|
159
|
+
# Handle both camelCase and snake_case response formats
|
|
160
|
+
call_ids = result.get("callExecutionIds") or result.get("call_execution_ids", [])
|
|
161
|
+
has_more = result.get("hasMore") if "hasMore" in result else result.get("has_more", False)
|
|
162
|
+
|
|
163
|
+
if not call_ids:
|
|
164
|
+
if has_more:
|
|
165
|
+
print("⚠️ Received empty batch but hasMore is true. Waiting...")
|
|
166
|
+
await asyncio.sleep(2)
|
|
167
|
+
continue
|
|
168
|
+
else:
|
|
169
|
+
break
|
|
170
|
+
|
|
171
|
+
print(f"📥 Received batch: {len(call_ids)} calls")
|
|
172
|
+
for cid in call_ids:
|
|
173
|
+
await queue.put(cid)
|
|
174
|
+
|
|
175
|
+
except Exception as e:
|
|
176
|
+
logger.error(f"Error fetching batch: {e}")
|
|
177
|
+
# Simple retry logic or break? For now, break to avoid infinite loop
|
|
178
|
+
break
|
|
179
|
+
|
|
180
|
+
def _simulator_baggage_context(self, call_execution_id: str):
|
|
181
|
+
"""
|
|
182
|
+
Creates a context manager that sets simulator IDs into OTEL baggage (via fi_instrumentation),
|
|
183
|
+
so any user-agent spans created inside the block inherit these attributes.
|
|
184
|
+
"""
|
|
185
|
+
if self._using_simulator_attributes is None:
|
|
186
|
+
return contextlib.nullcontext()
|
|
187
|
+
|
|
188
|
+
simulator_attributes = {
|
|
189
|
+
"is_simulator_trace": True,
|
|
190
|
+
"run_test_id": self.run_test_id,
|
|
191
|
+
"test_execution_id": self.test_execution_id,
|
|
192
|
+
"call_execution_id": call_execution_id,
|
|
193
|
+
}
|
|
194
|
+
# Remove None values to avoid serializing nulls
|
|
195
|
+
simulator_attributes = {k: v for k, v in simulator_attributes.items() if v is not None}
|
|
196
|
+
|
|
197
|
+
return self._using_simulator_attributes(simulator_attributes)
|
|
198
|
+
|
|
199
|
+
async def _consumer_loop(self, queue: asyncio.Queue, wrapper: AgentWrapper):
|
|
200
|
+
"""
|
|
201
|
+
Worker that pulls execution IDs from the queue and runs the conversation.
|
|
202
|
+
"""
|
|
203
|
+
while True:
|
|
204
|
+
try:
|
|
205
|
+
execution_id = await queue.get()
|
|
206
|
+
await self._handle_single_execution(execution_id, wrapper)
|
|
207
|
+
queue.task_done()
|
|
208
|
+
except asyncio.CancelledError:
|
|
209
|
+
break
|
|
210
|
+
except Exception as e:
|
|
211
|
+
error_msg = str(e) or f"{type(e).__name__}: {repr(e)}"
|
|
212
|
+
logger.error(f"Error in consumer: {error_msg}", exc_info=True)
|
|
213
|
+
print(f"❌ Consumer error: {error_msg}")
|
|
214
|
+
queue.task_done() # Mark done even if failed so join() works
|
|
215
|
+
|
|
216
|
+
async def _handle_single_execution(self, call_execution_id: str, wrapper: AgentWrapper):
|
|
217
|
+
"""
|
|
218
|
+
Runs the conversation loop for a single call execution.
|
|
219
|
+
"""
|
|
220
|
+
token = current_execution_id.set(call_execution_id)
|
|
221
|
+
try:
|
|
222
|
+
print(f"▶️ Processing Call: {call_execution_id}")
|
|
223
|
+
return await self._handle_single_execution_inner(call_execution_id, wrapper)
|
|
224
|
+
finally:
|
|
225
|
+
current_execution_id.reset(token)
|
|
226
|
+
|
|
227
|
+
async def _handle_single_execution_inner(self, call_execution_id: str, wrapper: AgentWrapper):
|
|
228
|
+
"""
|
|
229
|
+
Inner implementation of a single call execution. Separated so we can optionally wrap
|
|
230
|
+
the entire conversation in a parent tracing span and other instrumentation.
|
|
231
|
+
"""
|
|
232
|
+
try:
|
|
233
|
+
|
|
234
|
+
# Step 1: Initiate chat (POST with initiate_chat=True)
|
|
235
|
+
init_resp = await self.api.send_chat_message(
|
|
236
|
+
call_execution_id=call_execution_id,
|
|
237
|
+
initiate_chat=True
|
|
238
|
+
)
|
|
239
|
+
result = init_resp.get("result", {})
|
|
240
|
+
|
|
241
|
+
if not result:
|
|
242
|
+
logger.error(f"Failed to initiate chat for {call_execution_id}")
|
|
243
|
+
return
|
|
244
|
+
|
|
245
|
+
# Extract first message(s) from response
|
|
246
|
+
# Note: message_history is a list of ChatMessage objects (dicts)
|
|
247
|
+
message_history = result.get("message_history") or result.get("messageHistory", [])
|
|
248
|
+
|
|
249
|
+
if not message_history:
|
|
250
|
+
# Fallback to output_message if history is empty
|
|
251
|
+
output_msg = result.get("output_message") or result.get("outputMessage")
|
|
252
|
+
if output_msg:
|
|
253
|
+
# Ensure it's a list
|
|
254
|
+
if isinstance(output_msg, list):
|
|
255
|
+
message_history = output_msg
|
|
256
|
+
else:
|
|
257
|
+
message_history = [output_msg]
|
|
258
|
+
|
|
259
|
+
if not message_history:
|
|
260
|
+
logger.warning(f"No initial message received for {call_execution_id}")
|
|
261
|
+
return
|
|
262
|
+
|
|
263
|
+
# Build conversation history for SDK format
|
|
264
|
+
# Convert backend "assistant" → SDK "user" (simulator messages)
|
|
265
|
+
conversation_history = []
|
|
266
|
+
for msg in message_history:
|
|
267
|
+
backend_role = msg.get("role", "user")
|
|
268
|
+
|
|
269
|
+
# Filter out system and tool messages from backend (simulator artifacts)
|
|
270
|
+
if backend_role in ["system", "tool"]:
|
|
271
|
+
continue
|
|
272
|
+
|
|
273
|
+
# Filter out empty messages (often tool calls without output text yet)
|
|
274
|
+
content = msg.get("content", "")
|
|
275
|
+
if not content and backend_role == "assistant":
|
|
276
|
+
continue
|
|
277
|
+
|
|
278
|
+
# Backend sends simulator messages as "assistant", convert to "user" for SDK
|
|
279
|
+
sdk_role = "user" if backend_role == "assistant" else backend_role
|
|
280
|
+
conversation_history.append({
|
|
281
|
+
"role": sdk_role,
|
|
282
|
+
"content": content
|
|
283
|
+
})
|
|
284
|
+
|
|
285
|
+
# Step 2: Conversation loop
|
|
286
|
+
max_turns = 50 # Safety limit
|
|
287
|
+
turn_count = 0
|
|
288
|
+
agent_call_failed = False # Track if agent call failed
|
|
289
|
+
|
|
290
|
+
while turn_count < max_turns:
|
|
291
|
+
# Check if chat ended based on last response
|
|
292
|
+
chat_ended = result.get("chat_ended") or result.get("chatEnded", False)
|
|
293
|
+
if chat_ended:
|
|
294
|
+
break
|
|
295
|
+
|
|
296
|
+
# Get the last message (should be from simulator/user to reply to)
|
|
297
|
+
if not conversation_history:
|
|
298
|
+
break
|
|
299
|
+
|
|
300
|
+
last_msg = conversation_history[-1]
|
|
301
|
+
|
|
302
|
+
# Prepare AgentInput for user's wrapper
|
|
303
|
+
agent_input = AgentInput(
|
|
304
|
+
thread_id=call_execution_id,
|
|
305
|
+
messages=conversation_history,
|
|
306
|
+
new_message=last_msg,
|
|
307
|
+
execution_id=call_execution_id
|
|
308
|
+
)
|
|
309
|
+
|
|
310
|
+
# Call user's agent and measure latency
|
|
311
|
+
import time
|
|
312
|
+
start_time = time.time() # Fallback for latency calculation approximation
|
|
313
|
+
try:
|
|
314
|
+
# Propagate simulator IDs to any spans created by the user's agent instrumentation
|
|
315
|
+
with self._simulator_baggage_context(call_execution_id):
|
|
316
|
+
start_time = time.time() # Accurate start time for latency calculation
|
|
317
|
+
agent_response = await wrapper.call(agent_input)
|
|
318
|
+
except Exception as e:
|
|
319
|
+
error_msg = str(e) or f"{type(e).__name__}: {repr(e)}"
|
|
320
|
+
last_msg_content = agent_input.new_message.get('content', '') if agent_input.new_message else 'N/A'
|
|
321
|
+
logger.error(f"Agent call failed for {call_execution_id}: {error_msg}", exc_info=True)
|
|
322
|
+
print(f"❌ Agent call failed for {call_execution_id}: {error_msg}")
|
|
323
|
+
if last_msg_content != 'N/A':
|
|
324
|
+
print(f" Last message: {last_msg_content[:100]}...")
|
|
325
|
+
# Update call execution status in the backend
|
|
326
|
+
# If we have already completed some turns, mark as "completed" so evaluations can run
|
|
327
|
+
# on the partial data. Only mark as "failed" if we failed on the first turn.
|
|
328
|
+
status = "completed" if turn_count > 0 else "failed"
|
|
329
|
+
# Use generic error message to avoid leaking internal error details
|
|
330
|
+
generic_reason = "Error processing simulation"
|
|
331
|
+
try:
|
|
332
|
+
await self.api.update_call_execution_status(
|
|
333
|
+
call_execution_id,
|
|
334
|
+
status,
|
|
335
|
+
ended_reason=generic_reason
|
|
336
|
+
)
|
|
337
|
+
print(f" Status set to '{status}' (turn_count={turn_count})")
|
|
338
|
+
except Exception as status_error:
|
|
339
|
+
logger.warning(f"Failed to update call execution status for {call_execution_id}: {status_error}")
|
|
340
|
+
agent_call_failed = True
|
|
341
|
+
break
|
|
342
|
+
latency_ms = int((time.time() - start_time) * 1000) if start_time is not None else 0
|
|
343
|
+
|
|
344
|
+
# Normalize response and extract tool_calls and tool_responses
|
|
345
|
+
response_content = ""
|
|
346
|
+
tool_calls = None
|
|
347
|
+
tool_responses = None
|
|
348
|
+
|
|
349
|
+
if isinstance(agent_response, AgentResponse):
|
|
350
|
+
response_content = agent_response.content
|
|
351
|
+
tool_calls = agent_response.tool_calls
|
|
352
|
+
tool_responses = agent_response.tool_responses
|
|
353
|
+
# Back-compat: allow tool outputs to be passed via metadata["tool_outputs"]
|
|
354
|
+
# Expected shape: [{"call_id": "...", "output": ...}, ...]
|
|
355
|
+
if not tool_responses and agent_response.metadata:
|
|
356
|
+
tool_outputs = agent_response.metadata.get("tool_outputs")
|
|
357
|
+
if isinstance(tool_outputs, list) and tool_outputs:
|
|
358
|
+
import json
|
|
359
|
+
converted: list[dict] = []
|
|
360
|
+
for item in tool_outputs:
|
|
361
|
+
if not isinstance(item, dict):
|
|
362
|
+
continue
|
|
363
|
+
call_id = item.get("call_id") or item.get("tool_call_id")
|
|
364
|
+
output = item.get("output")
|
|
365
|
+
if call_id is None and output is None:
|
|
366
|
+
continue
|
|
367
|
+
converted.append(
|
|
368
|
+
{
|
|
369
|
+
"role": "tool",
|
|
370
|
+
"tool_call_id": call_id,
|
|
371
|
+
"content": output
|
|
372
|
+
if isinstance(output, str)
|
|
373
|
+
else json.dumps(output),
|
|
374
|
+
}
|
|
375
|
+
)
|
|
376
|
+
tool_responses = converted or None
|
|
377
|
+
else:
|
|
378
|
+
response_content = str(agent_response)
|
|
379
|
+
|
|
380
|
+
# Add agent response to history (with tool_calls if present)
|
|
381
|
+
assistant_msg = {
|
|
382
|
+
"role": "assistant",
|
|
383
|
+
"content": response_content
|
|
384
|
+
}
|
|
385
|
+
if tool_calls:
|
|
386
|
+
assistant_msg["tool_calls"] = tool_calls
|
|
387
|
+
conversation_history.append(assistant_msg)
|
|
388
|
+
|
|
389
|
+
# Add tool role messages (tool responses) after assistant message with tool_calls
|
|
390
|
+
if tool_responses:
|
|
391
|
+
for tool_response in tool_responses:
|
|
392
|
+
conversation_history.append(tool_response)
|
|
393
|
+
|
|
394
|
+
# Step 3: Send agent response to backend and get next message
|
|
395
|
+
# Send the assistant message with tool_calls and any tool responses
|
|
396
|
+
# SDK "assistant" (agent) → backend "user", SDK "tool" → backend "tool"
|
|
397
|
+
api_messages = []
|
|
398
|
+
|
|
399
|
+
# Add assistant message with tool_calls
|
|
400
|
+
assistant_api_msg = {
|
|
401
|
+
"role": "user", # Convert SDK "assistant" → backend "user"
|
|
402
|
+
"content": assistant_msg["content"]
|
|
403
|
+
}
|
|
404
|
+
if "tool_calls" in assistant_msg:
|
|
405
|
+
assistant_api_msg["tool_calls"] = assistant_msg["tool_calls"]
|
|
406
|
+
api_messages.append(assistant_api_msg)
|
|
407
|
+
|
|
408
|
+
# Add tool role messages if present
|
|
409
|
+
if tool_responses:
|
|
410
|
+
for tool_response in tool_responses:
|
|
411
|
+
api_messages.append({
|
|
412
|
+
"role": "tool", # Keep as "tool" for backend
|
|
413
|
+
"tool_call_id": tool_response.get("tool_call_id"),
|
|
414
|
+
"content": tool_response.get("content", "")
|
|
415
|
+
})
|
|
416
|
+
|
|
417
|
+
metrics = {"latency": latency_ms}
|
|
418
|
+
|
|
419
|
+
# Send
|
|
420
|
+
turn_resp = await self.api.send_chat_message(
|
|
421
|
+
call_execution_id=call_execution_id,
|
|
422
|
+
messages=api_messages,
|
|
423
|
+
metrics=metrics,
|
|
424
|
+
initiate_chat=False
|
|
425
|
+
)
|
|
426
|
+
|
|
427
|
+
result = turn_resp.get("result", {})
|
|
428
|
+
if not result:
|
|
429
|
+
logger.warning(f"No response from backend for {call_execution_id}")
|
|
430
|
+
break
|
|
431
|
+
|
|
432
|
+
# Update conversation history from backend response
|
|
433
|
+
|
|
434
|
+
new_history_data = result.get("message_history") or result.get("messageHistory", [])
|
|
435
|
+
|
|
436
|
+
if new_history_data:
|
|
437
|
+
# Convert backend "assistant" → SDK "user" (simulator messages)
|
|
438
|
+
conversation_history = []
|
|
439
|
+
for msg in new_history_data:
|
|
440
|
+
backend_role = msg.get("role", "user")
|
|
441
|
+
|
|
442
|
+
# Filter out system and tool messages
|
|
443
|
+
if backend_role in ["system", "tool"]:
|
|
444
|
+
continue
|
|
445
|
+
|
|
446
|
+
# Filter out empty messages
|
|
447
|
+
content = msg.get("content", "")
|
|
448
|
+
if not content and backend_role == "assistant":
|
|
449
|
+
continue
|
|
450
|
+
|
|
451
|
+
sdk_role = "user" if backend_role == "assistant" else backend_role
|
|
452
|
+
conversation_history.append({
|
|
453
|
+
"role": sdk_role,
|
|
454
|
+
"content": content
|
|
455
|
+
})
|
|
456
|
+
else:
|
|
457
|
+
# Fallback: append output_message if history missing
|
|
458
|
+
output_msgs = result.get("output_message") or result.get("outputMessage")
|
|
459
|
+
if output_msgs:
|
|
460
|
+
if isinstance(output_msgs, list):
|
|
461
|
+
for om in output_msgs:
|
|
462
|
+
backend_role = om.get("role", "user")
|
|
463
|
+
if backend_role in ["system", "tool"]:
|
|
464
|
+
continue
|
|
465
|
+
|
|
466
|
+
content = om.get("content", "")
|
|
467
|
+
if not content and backend_role == "assistant":
|
|
468
|
+
continue
|
|
469
|
+
|
|
470
|
+
sdk_role = "user" if backend_role == "assistant" else backend_role
|
|
471
|
+
conversation_history.append({
|
|
472
|
+
"role": sdk_role,
|
|
473
|
+
"content": content
|
|
474
|
+
})
|
|
475
|
+
else:
|
|
476
|
+
backend_role = output_msgs.get("role", "user")
|
|
477
|
+
if backend_role not in ["system", "tool"]:
|
|
478
|
+
content = output_msgs.get("content", "")
|
|
479
|
+
if content or backend_role != "assistant":
|
|
480
|
+
sdk_role = "user" if backend_role == "assistant" else backend_role
|
|
481
|
+
conversation_history.append({
|
|
482
|
+
"role": sdk_role,
|
|
483
|
+
"content": content
|
|
484
|
+
})
|
|
485
|
+
|
|
486
|
+
turn_count += 1
|
|
487
|
+
|
|
488
|
+
# Only print success if the call didn't fail
|
|
489
|
+
if not agent_call_failed:
|
|
490
|
+
print(f"✓ Call Finished: {call_execution_id} ({turn_count} turns)")
|
|
491
|
+
|
|
492
|
+
except Exception as e:
|
|
493
|
+
# Get detailed error message
|
|
494
|
+
error_msg = str(e)
|
|
495
|
+
if not error_msg:
|
|
496
|
+
error_msg = f"{type(e).__name__}: {repr(e)}"
|
|
497
|
+
|
|
498
|
+
# Log to both logger and console
|
|
499
|
+
logger.error(f"Call execution {call_execution_id} failed: {error_msg}", exc_info=True)
|
|
500
|
+
print(f"❌ Call execution {call_execution_id} failed: {error_msg}")
|
|
501
|
+
|
|
502
|
+
# Update call execution status to failed
|
|
503
|
+
try:
|
|
504
|
+
# Use "FAILED" (uppercase) to match Django model choices, and include error message as ended_reason
|
|
505
|
+
await self.api.update_call_execution_status(
|
|
506
|
+
call_execution_id,
|
|
507
|
+
"failed",
|
|
508
|
+
ended_reason=error_msg
|
|
509
|
+
)
|
|
510
|
+
except Exception as status_error:
|
|
511
|
+
# Don't let status update failure mask the original error
|
|
512
|
+
logger.warning(f"Failed to update call execution status for {call_execution_id}: {status_error}")
|
|
513
|
+
return None
|
|
514
|
+
|
|
515
|
+
def _normalize_callback(self, callback: Callable | AgentWrapper) -> AgentWrapper:
|
|
516
|
+
"""Ensures we have a AgentWrapper instance."""
|
|
517
|
+
return wrap_agent(callback)
|