agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
from typing import List, Dict, Union, Any, Optional, Literal
|
|
2
|
+
from pydantic import BaseModel, Field
|
|
3
|
+
from abc import ABC, abstractmethod
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
ArtifactType = Literal[
|
|
7
|
+
"text",
|
|
8
|
+
"image",
|
|
9
|
+
"audio",
|
|
10
|
+
"video",
|
|
11
|
+
"screenshot",
|
|
12
|
+
"browser_dom",
|
|
13
|
+
"file",
|
|
14
|
+
"json",
|
|
15
|
+
"trace",
|
|
16
|
+
]
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class SimulationArtifact(BaseModel):
|
|
20
|
+
"""
|
|
21
|
+
Modality-neutral artifact carried through a simulation.
|
|
22
|
+
|
|
23
|
+
Use `uri` or `path` for large media, `data` for small inline payloads, and
|
|
24
|
+
`metadata` for framework-specific details like sample rate, viewport, page
|
|
25
|
+
URL, or image dimensions.
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
type: ArtifactType
|
|
29
|
+
uri: Optional[str] = None
|
|
30
|
+
path: Optional[str] = None
|
|
31
|
+
data: Optional[Any] = None
|
|
32
|
+
mime_type: Optional[str] = None
|
|
33
|
+
role: Optional[str] = None
|
|
34
|
+
metadata: Dict[str, Any] = Field(default_factory=dict)
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
class SimulationEvent(BaseModel):
|
|
38
|
+
"""Normalized event for tools, memory, browser/CUA actions, voice states, and framework spans."""
|
|
39
|
+
|
|
40
|
+
type: str
|
|
41
|
+
name: Optional[str] = None
|
|
42
|
+
payload: Dict[str, Any] = Field(default_factory=dict)
|
|
43
|
+
timestamp_ms: Optional[int] = None
|
|
44
|
+
metadata: Dict[str, Any] = Field(default_factory=dict)
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
class AgentInput(BaseModel):
|
|
48
|
+
"""
|
|
49
|
+
Input data passed to the user's agent wrapper during a simulation step.
|
|
50
|
+
"""
|
|
51
|
+
thread_id: str
|
|
52
|
+
messages: List[Dict[str, Any]] # Full conversation history: [{"role": "user", "content": "..."}]
|
|
53
|
+
new_message: Optional[Dict[str, Any]] = None # The latest message to respond to
|
|
54
|
+
|
|
55
|
+
# Metadata for execution context (useful for logging/debugging)
|
|
56
|
+
execution_id: Optional[str] = None
|
|
57
|
+
turn_index: Optional[int] = None
|
|
58
|
+
scenario_name: Optional[str] = None
|
|
59
|
+
persona: Optional[Dict[str, Any]] = None
|
|
60
|
+
situation: Optional[str] = None
|
|
61
|
+
expected_outcome: Optional[str] = None
|
|
62
|
+
modality: Optional[str] = None
|
|
63
|
+
artifacts: List[SimulationArtifact] = Field(default_factory=list)
|
|
64
|
+
events: List[SimulationEvent] = Field(default_factory=list)
|
|
65
|
+
memory: Dict[str, Any] = Field(default_factory=dict)
|
|
66
|
+
tools: List[Dict[str, Any]] = Field(default_factory=list)
|
|
67
|
+
metadata: Dict[str, Any] = Field(default_factory=dict)
|
|
68
|
+
|
|
69
|
+
class AgentResponse(BaseModel):
|
|
70
|
+
"""
|
|
71
|
+
Standardized response from the user's agent.
|
|
72
|
+
"""
|
|
73
|
+
content: str
|
|
74
|
+
tool_calls: Optional[List[Dict[str, Any]]] = None
|
|
75
|
+
tool_responses: Optional[List[Dict[str, Any]]] = None # Tool role messages with results
|
|
76
|
+
artifacts: List[SimulationArtifact] = Field(default_factory=list)
|
|
77
|
+
events: List[SimulationEvent] = Field(default_factory=list)
|
|
78
|
+
memory_updates: Optional[Dict[str, Any]] = None
|
|
79
|
+
state: Optional[Dict[str, Any]] = None
|
|
80
|
+
metadata: Optional[Dict[str, Any]] = None
|
|
81
|
+
|
|
82
|
+
class AgentWrapper(ABC):
|
|
83
|
+
"""
|
|
84
|
+
Base class for wrapping user agents to work with the simulation SDK.
|
|
85
|
+
Users should implement the `call` method.
|
|
86
|
+
"""
|
|
87
|
+
|
|
88
|
+
@abstractmethod
|
|
89
|
+
async def call(self, input: AgentInput) -> Union[str, AgentResponse]:
|
|
90
|
+
"""
|
|
91
|
+
Process the input and return the agent's response.
|
|
92
|
+
|
|
93
|
+
Args:
|
|
94
|
+
input: The AgentInput object containing message history and context.
|
|
95
|
+
|
|
96
|
+
Returns:
|
|
97
|
+
A string (content only) or AgentResponse object.
|
|
98
|
+
"""
|
|
99
|
+
pass
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
from fi.simulate.agent.wrappers.openai import OpenAIAgentWrapper
|
|
2
|
+
from fi.simulate.agent.wrappers.langchain import LangChainAgentWrapper
|
|
3
|
+
from fi.simulate.agent.wrappers.gemini import GeminiAgentWrapper
|
|
4
|
+
from fi.simulate.agent.wrappers.anthropic import AnthropicAgentWrapper
|
|
5
|
+
from fi.simulate.agent.wrappers.http import HTTPAgentWrapper
|
|
6
|
+
from fi.simulate.agent.wrappers.websocket import WebSocketAgentWrapper
|
|
7
|
+
|
|
8
|
+
OpenAICompatibleHTTPAgentWrapper = HTTPAgentWrapper
|
|
9
|
+
|
|
10
|
+
__all__ = [
|
|
11
|
+
"OpenAIAgentWrapper",
|
|
12
|
+
"LangChainAgentWrapper",
|
|
13
|
+
"GeminiAgentWrapper",
|
|
14
|
+
"AnthropicAgentWrapper",
|
|
15
|
+
"HTTPAgentWrapper",
|
|
16
|
+
"OpenAICompatibleHTTPAgentWrapper",
|
|
17
|
+
"WebSocketAgentWrapper",
|
|
18
|
+
]
|
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
from typing import Any, Union
|
|
2
|
+
from fi.simulate.agent.wrapper import AgentWrapper, AgentInput, AgentResponse
|
|
3
|
+
|
|
4
|
+
class AnthropicAgentWrapper(AgentWrapper):
|
|
5
|
+
"""
|
|
6
|
+
Wrapper for Anthropic (Claude) agents.
|
|
7
|
+
Automatically handles message conversion to Anthropic format.
|
|
8
|
+
"""
|
|
9
|
+
def __init__(self, client: Any, model: str = "claude-sonnet-4-5-20250929", system_prompt: str = None, max_tokens: int = 1024):
|
|
10
|
+
"""
|
|
11
|
+
Args:
|
|
12
|
+
client: The Anthropic client instance (AsyncAnthropic or Anthropic).
|
|
13
|
+
model: The model name to use.
|
|
14
|
+
system_prompt: Optional system instructions for the agent.
|
|
15
|
+
max_tokens: Maximum number of tokens to generate (default: 1024).
|
|
16
|
+
"""
|
|
17
|
+
self.client = client
|
|
18
|
+
self.model = model
|
|
19
|
+
self.system_prompt = system_prompt
|
|
20
|
+
self.max_tokens = max_tokens
|
|
21
|
+
|
|
22
|
+
async def call(self, input: AgentInput) -> Union[str, AgentResponse]:
|
|
23
|
+
# Convert internal message format to Anthropic format
|
|
24
|
+
# Anthropic messages API expects: [{"role": "user"|"assistant", "content": "..."}]
|
|
25
|
+
# It does NOT support "system" role in the messages list; system prompt is a top-level param.
|
|
26
|
+
|
|
27
|
+
messages = []
|
|
28
|
+
# Use configured system prompt by default
|
|
29
|
+
system_prompt = self.system_prompt
|
|
30
|
+
|
|
31
|
+
for msg in input.messages:
|
|
32
|
+
if msg["role"] == "system":
|
|
33
|
+
# If history has system message (unlikely due to filtering), it overrides?
|
|
34
|
+
# Or we ignore it to respect wrapper config?
|
|
35
|
+
# Let's check if it exists and use it if self.system_prompt is None
|
|
36
|
+
if system_prompt is None:
|
|
37
|
+
system_prompt = msg["content"]
|
|
38
|
+
else:
|
|
39
|
+
messages.append({
|
|
40
|
+
"role": msg["role"],
|
|
41
|
+
"content": msg["content"]
|
|
42
|
+
})
|
|
43
|
+
|
|
44
|
+
# Check for AsyncAnthropic vs Sync
|
|
45
|
+
# Heuristic: check for 'messages.create' and if client class name contains Async
|
|
46
|
+
is_async = type(self.client).__name__.startswith("Async")
|
|
47
|
+
|
|
48
|
+
kwargs = {
|
|
49
|
+
"model": self.model,
|
|
50
|
+
"max_tokens": self.max_tokens,
|
|
51
|
+
"messages": messages
|
|
52
|
+
}
|
|
53
|
+
if system_prompt:
|
|
54
|
+
kwargs["system"] = system_prompt
|
|
55
|
+
|
|
56
|
+
if is_async:
|
|
57
|
+
message = await self.client.messages.create(**kwargs)
|
|
58
|
+
else:
|
|
59
|
+
message = self.client.messages.create(**kwargs)
|
|
60
|
+
|
|
61
|
+
return message.content[0].text
|
|
62
|
+
|
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
from typing import Any, Union
|
|
2
|
+
from fi.simulate.agent.wrapper import AgentWrapper, AgentInput, AgentResponse
|
|
3
|
+
|
|
4
|
+
class GeminiAgentWrapper(AgentWrapper):
|
|
5
|
+
"""
|
|
6
|
+
Wrapper for Google Gemini (Generative AI) agents.
|
|
7
|
+
Supports google-generativeai SDK.
|
|
8
|
+
"""
|
|
9
|
+
def __init__(self, model: Any, system_prompt: str = None):
|
|
10
|
+
"""
|
|
11
|
+
Args:
|
|
12
|
+
model: An instance of google.generativeai.GenerativeModel
|
|
13
|
+
system_prompt: Optional system instructions.
|
|
14
|
+
Note: Ideally configure system_instruction on the model itself.
|
|
15
|
+
If provided here, it will be prepended as a user message.
|
|
16
|
+
"""
|
|
17
|
+
self.model = model
|
|
18
|
+
self.system_prompt = system_prompt
|
|
19
|
+
|
|
20
|
+
async def call(self, input: AgentInput) -> Union[str, AgentResponse]:
|
|
21
|
+
# Convert internal messages to Gemini format (Content objects)
|
|
22
|
+
# Note: Gemini SDK manages chat history via ChatSession usually,
|
|
23
|
+
# but for stateless call we pass full history if supported,
|
|
24
|
+
# or we might need to reconstruct a chat session.
|
|
25
|
+
|
|
26
|
+
# Simple reconstruction of history for a chat session
|
|
27
|
+
history = []
|
|
28
|
+
|
|
29
|
+
if self.system_prompt:
|
|
30
|
+
# Prepend system prompt as a user message for context
|
|
31
|
+
history.append({"role": "user", "parts": [f"System Instruction: {self.system_prompt}"]})
|
|
32
|
+
# Add a dummy model acknowledgement to keep turns valid (User -> Model -> User)
|
|
33
|
+
history.append({"role": "model", "parts": ["Understood."]})
|
|
34
|
+
|
|
35
|
+
for msg in input.messages:
|
|
36
|
+
role = "user" if msg["role"] == "user" else "model"
|
|
37
|
+
content = msg["content"]
|
|
38
|
+
|
|
39
|
+
# Gemini typically expects history excluding the last message which is passed to send_message
|
|
40
|
+
history.append({"role": role, "parts": [content]})
|
|
41
|
+
|
|
42
|
+
if not history:
|
|
43
|
+
raise ValueError("No messages provided to Gemini wrapper")
|
|
44
|
+
|
|
45
|
+
# The last user message is the prompt
|
|
46
|
+
last_turn = history.pop()
|
|
47
|
+
if last_turn["role"] != "user":
|
|
48
|
+
# If the last message wasn't user, something is weird in the flow,
|
|
49
|
+
# but we can try to send empty or handle it.
|
|
50
|
+
# Ideally simulator sends User message last.
|
|
51
|
+
prompt = ""
|
|
52
|
+
else:
|
|
53
|
+
prompt = last_turn["parts"][0]
|
|
54
|
+
|
|
55
|
+
# Start a chat with the history
|
|
56
|
+
chat = self.model.start_chat(history=history)
|
|
57
|
+
|
|
58
|
+
# Check if async generation is supported (google-generativeai >= 0.3.0 has send_message_async)
|
|
59
|
+
if hasattr(chat, "send_message_async"):
|
|
60
|
+
response = await chat.send_message_async(prompt)
|
|
61
|
+
else:
|
|
62
|
+
# Fallback to sync
|
|
63
|
+
response = chat.send_message(prompt)
|
|
64
|
+
|
|
65
|
+
return response.text
|
|
@@ -0,0 +1,404 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import asyncio
|
|
4
|
+
import json
|
|
5
|
+
import os
|
|
6
|
+
import time
|
|
7
|
+
import urllib.error
|
|
8
|
+
import urllib.request
|
|
9
|
+
from typing import Any, Mapping, Optional, Sequence
|
|
10
|
+
from urllib.parse import urlparse
|
|
11
|
+
|
|
12
|
+
from fi.simulate.agent.wrapper import (
|
|
13
|
+
AgentInput,
|
|
14
|
+
AgentResponse,
|
|
15
|
+
SimulationArtifact,
|
|
16
|
+
SimulationEvent,
|
|
17
|
+
)
|
|
18
|
+
from fi.simulate.agent.wrapper import AgentWrapper
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class HTTPAgentWrapper(AgentWrapper):
|
|
22
|
+
"""HTTP/OpenAI-compatible target adapter for external agent simulation."""
|
|
23
|
+
|
|
24
|
+
def __init__(
|
|
25
|
+
self,
|
|
26
|
+
*,
|
|
27
|
+
endpoint: str,
|
|
28
|
+
protocol: str = "fi.alk",
|
|
29
|
+
model: Optional[str] = None,
|
|
30
|
+
api_key: Optional[str] = None,
|
|
31
|
+
api_key_env: Optional[str] = None,
|
|
32
|
+
headers: Optional[Mapping[str, str]] = None,
|
|
33
|
+
timeout: float = 30.0,
|
|
34
|
+
include_tools: bool = True,
|
|
35
|
+
system_prompt: Optional[str] = None,
|
|
36
|
+
metadata: Optional[Mapping[str, Any]] = None,
|
|
37
|
+
) -> None:
|
|
38
|
+
if not endpoint:
|
|
39
|
+
raise ValueError("endpoint is required")
|
|
40
|
+
self.endpoint = endpoint
|
|
41
|
+
self.protocol = _normalize_protocol(protocol)
|
|
42
|
+
self.model = model
|
|
43
|
+
self.api_key = api_key
|
|
44
|
+
self.api_key_env = api_key_env
|
|
45
|
+
self.headers = {str(k): str(v) for k, v in dict(headers or {}).items()}
|
|
46
|
+
self.timeout = float(timeout)
|
|
47
|
+
self.include_tools = bool(include_tools)
|
|
48
|
+
self.system_prompt = system_prompt
|
|
49
|
+
self.metadata = dict(metadata or {})
|
|
50
|
+
|
|
51
|
+
async def call(self, input: AgentInput) -> AgentResponse:
|
|
52
|
+
started = time.time()
|
|
53
|
+
request_payload = self._request_payload(input)
|
|
54
|
+
headers = self._request_headers()
|
|
55
|
+
status_code = 0
|
|
56
|
+
response_payload: dict[str, Any] = {}
|
|
57
|
+
error: Optional[str] = None
|
|
58
|
+
try:
|
|
59
|
+
status_code, response_payload = await asyncio.to_thread(
|
|
60
|
+
self._post_json,
|
|
61
|
+
request_payload,
|
|
62
|
+
headers,
|
|
63
|
+
)
|
|
64
|
+
if status_code >= 400:
|
|
65
|
+
error = _response_error_text(response_payload) or (
|
|
66
|
+
f"HTTP target returned status {status_code}"
|
|
67
|
+
)
|
|
68
|
+
response = self._agent_response_from_payload(response_payload)
|
|
69
|
+
except Exception as exc:
|
|
70
|
+
error = str(exc)
|
|
71
|
+
response = AgentResponse(content=f"HTTP target failed: {exc}")
|
|
72
|
+
|
|
73
|
+
latency_ms = round((time.time() - started) * 1000, 4)
|
|
74
|
+
trace = {
|
|
75
|
+
"kind": "external_agent_http_trace",
|
|
76
|
+
"protocol": self.protocol,
|
|
77
|
+
"endpoint": _redacted_endpoint(self.endpoint),
|
|
78
|
+
"endpoint_host": urlparse(self.endpoint).netloc,
|
|
79
|
+
"model": self.model,
|
|
80
|
+
"status_code": status_code,
|
|
81
|
+
"latency_ms": latency_ms,
|
|
82
|
+
"request_message_count": len(input.messages),
|
|
83
|
+
"request_tool_count": len(input.tools) if self.include_tools else 0,
|
|
84
|
+
"response_tool_call_count": len(response.tool_calls or []),
|
|
85
|
+
"success": error is None and 200 <= status_code < 300,
|
|
86
|
+
"request_header_names": sorted(headers),
|
|
87
|
+
"auth": {
|
|
88
|
+
"mode": "bearer" if self._resolved_api_key() else "none",
|
|
89
|
+
"api_key_env": self.api_key_env,
|
|
90
|
+
"redacted": bool(self._resolved_api_key()),
|
|
91
|
+
},
|
|
92
|
+
"error": error,
|
|
93
|
+
**self.metadata,
|
|
94
|
+
}
|
|
95
|
+
response.events.append(
|
|
96
|
+
SimulationEvent(
|
|
97
|
+
type="external_agent",
|
|
98
|
+
name="external_agent_http_call",
|
|
99
|
+
payload=trace,
|
|
100
|
+
)
|
|
101
|
+
)
|
|
102
|
+
response.artifacts.append(
|
|
103
|
+
SimulationArtifact(
|
|
104
|
+
type="trace",
|
|
105
|
+
role="agent",
|
|
106
|
+
data=trace,
|
|
107
|
+
metadata={"kind": "external_agent_http_trace"},
|
|
108
|
+
)
|
|
109
|
+
)
|
|
110
|
+
state = dict(response.state or {})
|
|
111
|
+
state["external_agent"] = trace
|
|
112
|
+
state["external_agent_trace"] = trace
|
|
113
|
+
response.state = state
|
|
114
|
+
metadata = dict(response.metadata or {})
|
|
115
|
+
metadata["external_agent"] = trace
|
|
116
|
+
metadata["external_agent_trace"] = trace
|
|
117
|
+
response.metadata = metadata
|
|
118
|
+
return response
|
|
119
|
+
|
|
120
|
+
def _request_payload(self, input: AgentInput) -> dict[str, Any]:
|
|
121
|
+
messages = _messages_for_protocol(input.messages, self.protocol)
|
|
122
|
+
if self.system_prompt:
|
|
123
|
+
messages = [{"role": "system", "content": self.system_prompt}, *messages]
|
|
124
|
+
if self.protocol == "openai_chat":
|
|
125
|
+
payload: dict[str, Any] = {
|
|
126
|
+
"model": self.model or "agent-learning-target",
|
|
127
|
+
"messages": messages,
|
|
128
|
+
}
|
|
129
|
+
if self.include_tools and input.tools:
|
|
130
|
+
payload["tools"] = [_openai_tool_spec(tool) for tool in input.tools]
|
|
131
|
+
payload["tool_choice"] = "auto"
|
|
132
|
+
return payload
|
|
133
|
+
return {
|
|
134
|
+
"thread_id": input.thread_id,
|
|
135
|
+
"execution_id": input.execution_id,
|
|
136
|
+
"turn_index": input.turn_index,
|
|
137
|
+
"scenario_name": input.scenario_name,
|
|
138
|
+
"persona": input.persona,
|
|
139
|
+
"situation": input.situation,
|
|
140
|
+
"expected_outcome": input.expected_outcome,
|
|
141
|
+
"messages": messages,
|
|
142
|
+
"new_message": input.new_message,
|
|
143
|
+
"tools": list(input.tools) if self.include_tools else [],
|
|
144
|
+
"metadata": input.metadata,
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
def _request_headers(self) -> dict[str, str]:
|
|
148
|
+
headers = {"Content-Type": "application/json", **self.headers}
|
|
149
|
+
api_key = self._resolved_api_key()
|
|
150
|
+
if api_key and not any(key.lower() == "authorization" for key in headers):
|
|
151
|
+
headers["Authorization"] = f"Bearer {api_key}"
|
|
152
|
+
return headers
|
|
153
|
+
|
|
154
|
+
def _resolved_api_key(self) -> str:
|
|
155
|
+
if self.api_key not in (None, ""):
|
|
156
|
+
return str(self.api_key)
|
|
157
|
+
if self.api_key_env:
|
|
158
|
+
return os.environ.get(self.api_key_env, "")
|
|
159
|
+
return ""
|
|
160
|
+
|
|
161
|
+
def _post_json(
|
|
162
|
+
self,
|
|
163
|
+
payload: Mapping[str, Any],
|
|
164
|
+
headers: Mapping[str, str],
|
|
165
|
+
) -> tuple[int, dict[str, Any]]:
|
|
166
|
+
body = json.dumps(payload, default=str).encode("utf-8")
|
|
167
|
+
request = urllib.request.Request(
|
|
168
|
+
self.endpoint,
|
|
169
|
+
data=body,
|
|
170
|
+
headers=dict(headers),
|
|
171
|
+
method="POST",
|
|
172
|
+
)
|
|
173
|
+
try:
|
|
174
|
+
with urllib.request.urlopen(request, timeout=self.timeout) as response:
|
|
175
|
+
status = int(getattr(response, "status", 200))
|
|
176
|
+
text = response.read().decode("utf-8")
|
|
177
|
+
except urllib.error.HTTPError as exc:
|
|
178
|
+
status = int(exc.code)
|
|
179
|
+
text = exc.read().decode("utf-8")
|
|
180
|
+
if not text:
|
|
181
|
+
return status, {}
|
|
182
|
+
try:
|
|
183
|
+
parsed = json.loads(text)
|
|
184
|
+
except json.JSONDecodeError as exc:
|
|
185
|
+
raise ValueError(f"HTTP target returned non-JSON response: {exc}") from exc
|
|
186
|
+
if not isinstance(parsed, dict):
|
|
187
|
+
raise ValueError("HTTP target response must be a JSON object")
|
|
188
|
+
return status, parsed
|
|
189
|
+
|
|
190
|
+
def _agent_response_from_payload(self, payload: Mapping[str, Any]) -> AgentResponse:
|
|
191
|
+
if self.protocol == "openai_chat":
|
|
192
|
+
message = _openai_message(payload)
|
|
193
|
+
return AgentResponse(
|
|
194
|
+
content=_content_text(message.get("content")),
|
|
195
|
+
tool_calls=_openai_tool_calls(message.get("tool_calls")),
|
|
196
|
+
metadata={
|
|
197
|
+
"finish_reason": _openai_finish_reason(payload),
|
|
198
|
+
"usage": dict(payload.get("usage") or {}),
|
|
199
|
+
},
|
|
200
|
+
)
|
|
201
|
+
return AgentResponse(
|
|
202
|
+
content=_content_text(payload.get("content") or payload.get("message")),
|
|
203
|
+
tool_calls=_tool_call_list(payload.get("tool_calls")),
|
|
204
|
+
tool_responses=_tool_response_list(payload.get("tool_responses")),
|
|
205
|
+
artifacts=_artifact_list(payload.get("artifacts")),
|
|
206
|
+
events=_event_list(payload.get("events")),
|
|
207
|
+
memory_updates=_optional_mapping(payload.get("memory_updates")),
|
|
208
|
+
state=_optional_mapping(payload.get("state")),
|
|
209
|
+
metadata=_optional_mapping(payload.get("metadata")),
|
|
210
|
+
)
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
def _normalize_protocol(value: str) -> str:
|
|
214
|
+
protocol = str(value or "fi.alk").lower().replace("-", "_")
|
|
215
|
+
aliases = {
|
|
216
|
+
"openai": "openai_chat",
|
|
217
|
+
"openai_compatible": "openai_chat",
|
|
218
|
+
"chat_completions": "openai_chat",
|
|
219
|
+
"agent_learning_http": "fi.alk",
|
|
220
|
+
"http": "fi.alk",
|
|
221
|
+
}
|
|
222
|
+
protocol = aliases.get(protocol, protocol)
|
|
223
|
+
if protocol not in {"fi.alk", "openai_chat"}:
|
|
224
|
+
raise ValueError("protocol must be one of: fi.alk, openai_chat")
|
|
225
|
+
return protocol
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
def _messages_for_protocol(
|
|
229
|
+
messages: Sequence[Mapping[str, Any]], protocol: str
|
|
230
|
+
) -> list[dict[str, Any]]:
|
|
231
|
+
"""Encode canonical ALK history for the selected external-agent protocol.
|
|
232
|
+
|
|
233
|
+
ALK keeps tool calls structured internally. OpenAI-compatible endpoints require that array
|
|
234
|
+
unchanged, while the FutureAGI callback ``AgentInput`` schema represents historical
|
|
235
|
+
``tool_calls`` as a JSON string. Normalizing here keeps the conversation runner generic and
|
|
236
|
+
prevents a retry from accidentally executing a side-effecting tool twice.
|
|
237
|
+
"""
|
|
238
|
+
normalized = [dict(message) for message in messages]
|
|
239
|
+
if protocol != "fi.alk":
|
|
240
|
+
return normalized
|
|
241
|
+
for message in normalized:
|
|
242
|
+
tool_calls = message.get("tool_calls")
|
|
243
|
+
if isinstance(tool_calls, Sequence) and not isinstance(
|
|
244
|
+
tool_calls, (str, bytes)
|
|
245
|
+
):
|
|
246
|
+
message["tool_calls"] = json.dumps(
|
|
247
|
+
list(tool_calls), separators=(",", ":"), default=str
|
|
248
|
+
)
|
|
249
|
+
return normalized
|
|
250
|
+
|
|
251
|
+
|
|
252
|
+
def _openai_tool_spec(tool: Mapping[str, Any]) -> dict[str, Any]:
|
|
253
|
+
# Accept both the flat SDK tool shape ({name, description, parameters}) and
|
|
254
|
+
# the OpenAI-nested shape ({"type": "function", "function": {...}}). Without
|
|
255
|
+
# reading the nested ``function`` block, a nested spec loses its name and the
|
|
256
|
+
# model is handed a tool literally called "tool" — so it can never call the
|
|
257
|
+
# real tool and the environment's mock never matches.
|
|
258
|
+
fn = tool.get("function") if isinstance(tool.get("function"), Mapping) else {}
|
|
259
|
+
name = str(
|
|
260
|
+
tool.get("name")
|
|
261
|
+
or fn.get("name")
|
|
262
|
+
or tool.get("tool")
|
|
263
|
+
or tool.get("id")
|
|
264
|
+
or "tool"
|
|
265
|
+
)
|
|
266
|
+
parameters = tool.get("parameters")
|
|
267
|
+
if not isinstance(parameters, Mapping):
|
|
268
|
+
parameters = fn.get("parameters")
|
|
269
|
+
if not isinstance(parameters, Mapping):
|
|
270
|
+
parameters = {"type": "object", "properties": {}}
|
|
271
|
+
description = tool.get("description") or fn.get("description") or f"Tool {name}"
|
|
272
|
+
return {
|
|
273
|
+
"type": "function",
|
|
274
|
+
"function": {
|
|
275
|
+
"name": name,
|
|
276
|
+
"description": str(description),
|
|
277
|
+
"parameters": dict(parameters),
|
|
278
|
+
},
|
|
279
|
+
}
|
|
280
|
+
|
|
281
|
+
|
|
282
|
+
def _openai_message(payload: Mapping[str, Any]) -> dict[str, Any]:
|
|
283
|
+
choices = payload.get("choices")
|
|
284
|
+
if isinstance(choices, Sequence) and not isinstance(choices, (str, bytes)):
|
|
285
|
+
if choices:
|
|
286
|
+
choice = choices[0]
|
|
287
|
+
if isinstance(choice, Mapping):
|
|
288
|
+
message = choice.get("message")
|
|
289
|
+
if isinstance(message, Mapping):
|
|
290
|
+
return dict(message)
|
|
291
|
+
message = payload.get("message")
|
|
292
|
+
return dict(message) if isinstance(message, Mapping) else dict(payload)
|
|
293
|
+
|
|
294
|
+
|
|
295
|
+
def _openai_finish_reason(payload: Mapping[str, Any]) -> Optional[str]:
|
|
296
|
+
choices = payload.get("choices")
|
|
297
|
+
if isinstance(choices, Sequence) and not isinstance(choices, (str, bytes)):
|
|
298
|
+
if choices and isinstance(choices[0], Mapping):
|
|
299
|
+
value = choices[0].get("finish_reason")
|
|
300
|
+
return str(value) if value is not None else None
|
|
301
|
+
return None
|
|
302
|
+
|
|
303
|
+
|
|
304
|
+
def _openai_tool_calls(value: Any) -> list[dict[str, Any]]:
|
|
305
|
+
calls = _tool_call_list(value)
|
|
306
|
+
normalized: list[dict[str, Any]] = []
|
|
307
|
+
for index, call in enumerate(calls, start=1):
|
|
308
|
+
function = call.get("function")
|
|
309
|
+
if isinstance(function, Mapping):
|
|
310
|
+
name = function.get("name")
|
|
311
|
+
arguments = function.get("arguments", {})
|
|
312
|
+
else:
|
|
313
|
+
name = call.get("name") or call.get("tool")
|
|
314
|
+
arguments = call.get("arguments", call.get("args", {}))
|
|
315
|
+
normalized.append(
|
|
316
|
+
{
|
|
317
|
+
"id": str(call.get("id") or f"call_{index}"),
|
|
318
|
+
"type": str(call.get("type") or "function"),
|
|
319
|
+
"function": {
|
|
320
|
+
"name": str(name or ""),
|
|
321
|
+
"arguments": (
|
|
322
|
+
arguments
|
|
323
|
+
if isinstance(arguments, str)
|
|
324
|
+
else json.dumps(arguments or {}, default=str)
|
|
325
|
+
),
|
|
326
|
+
},
|
|
327
|
+
}
|
|
328
|
+
)
|
|
329
|
+
return normalized
|
|
330
|
+
|
|
331
|
+
|
|
332
|
+
def _tool_call_list(value: Any) -> list[dict[str, Any]]:
|
|
333
|
+
if not isinstance(value, Sequence) or isinstance(value, (str, bytes)):
|
|
334
|
+
return []
|
|
335
|
+
return [dict(item) for item in value if isinstance(item, Mapping)]
|
|
336
|
+
|
|
337
|
+
|
|
338
|
+
def _tool_response_list(value: Any) -> list[dict[str, Any]]:
|
|
339
|
+
if not isinstance(value, Sequence) or isinstance(value, (str, bytes)):
|
|
340
|
+
return []
|
|
341
|
+
return [dict(item) for item in value if isinstance(item, Mapping)]
|
|
342
|
+
|
|
343
|
+
|
|
344
|
+
def _artifact_list(value: Any) -> list[SimulationArtifact]:
|
|
345
|
+
if not isinstance(value, Sequence) or isinstance(value, (str, bytes)):
|
|
346
|
+
return []
|
|
347
|
+
artifacts: list[SimulationArtifact] = []
|
|
348
|
+
for item in value:
|
|
349
|
+
if isinstance(item, SimulationArtifact):
|
|
350
|
+
artifacts.append(item)
|
|
351
|
+
elif isinstance(item, Mapping):
|
|
352
|
+
artifacts.append(SimulationArtifact(**dict(item)))
|
|
353
|
+
return artifacts
|
|
354
|
+
|
|
355
|
+
|
|
356
|
+
def _event_list(value: Any) -> list[SimulationEvent]:
|
|
357
|
+
if not isinstance(value, Sequence) or isinstance(value, (str, bytes)):
|
|
358
|
+
return []
|
|
359
|
+
events: list[SimulationEvent] = []
|
|
360
|
+
for item in value:
|
|
361
|
+
if isinstance(item, SimulationEvent):
|
|
362
|
+
events.append(item)
|
|
363
|
+
elif isinstance(item, Mapping):
|
|
364
|
+
events.append(SimulationEvent(**dict(item)))
|
|
365
|
+
return events
|
|
366
|
+
|
|
367
|
+
|
|
368
|
+
def _optional_mapping(value: Any) -> Optional[dict[str, Any]]:
|
|
369
|
+
return dict(value) if isinstance(value, Mapping) else None
|
|
370
|
+
|
|
371
|
+
|
|
372
|
+
def _content_text(value: Any) -> str:
|
|
373
|
+
if isinstance(value, str):
|
|
374
|
+
return value
|
|
375
|
+
if isinstance(value, Sequence) and not isinstance(value, (str, bytes)):
|
|
376
|
+
parts: list[str] = []
|
|
377
|
+
for item in value:
|
|
378
|
+
if isinstance(item, Mapping):
|
|
379
|
+
text = item.get("text") or item.get("content") or item.get("refusal")
|
|
380
|
+
if text not in (None, ""):
|
|
381
|
+
parts.append(str(text))
|
|
382
|
+
elif item not in (None, ""):
|
|
383
|
+
parts.append(str(item))
|
|
384
|
+
return "\n".join(parts)
|
|
385
|
+
return "" if value is None else str(value)
|
|
386
|
+
|
|
387
|
+
|
|
388
|
+
def _response_error_text(payload: Mapping[str, Any]) -> str:
|
|
389
|
+
error = payload.get("error")
|
|
390
|
+
if isinstance(error, Mapping):
|
|
391
|
+
return _content_text(error.get("message") or error.get("detail") or error)
|
|
392
|
+
if error not in (None, ""):
|
|
393
|
+
return _content_text(error)
|
|
394
|
+
for key in ("detail", "message", "status"):
|
|
395
|
+
if payload.get(key) not in (None, ""):
|
|
396
|
+
return _content_text(payload.get(key))
|
|
397
|
+
return ""
|
|
398
|
+
|
|
399
|
+
|
|
400
|
+
def _redacted_endpoint(endpoint: str) -> str:
|
|
401
|
+
parsed = urlparse(endpoint)
|
|
402
|
+
if not parsed.query:
|
|
403
|
+
return endpoint
|
|
404
|
+
return parsed._replace(query="<redacted>").geturl()
|