agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
fi/alk/voice_loop.py
ADDED
|
@@ -0,0 +1,174 @@
|
|
|
1
|
+
"""Phase 9A unit 4 — the voice improvement loop (the 13D Practice Loop on
|
|
2
|
+
``world.kind = voice_telephony``).
|
|
3
|
+
|
|
4
|
+
ARCH §2.3 / decisions 9A-D5 (the 13D loop on voice_telephony; NO new optimizer),
|
|
5
|
+
9A-A4 (multi-objective-mandatory voice loss; single-timing-term rejection),
|
|
6
|
+
9A-A14 (``V1_VOICE_FAILURE_SUBLAYERS`` voice sub-attribution).
|
|
7
|
+
|
|
8
|
+
This module invents NO optimizer, NO artifact kind, NO loss machinery. It is a
|
|
9
|
+
thin composition layer over verbatim engines:
|
|
10
|
+
|
|
11
|
+
* the multi-objective voice loss compiles via ``loss.compile_objective`` (the
|
|
12
|
+
Goodhart guard at ``loss.py:106-116`` is reused VERBATIM — "There is no
|
|
13
|
+
override."); the 9A-A4 composition rule (≥2 terms, ≥1 non-timing quality
|
|
14
|
+
term) is a thin validator on top, raising ``voice_loss_guard_missing``;
|
|
15
|
+
* the whole voice-agent config is the search space, assembled by
|
|
16
|
+
``optimize.build_practice_loop_manifest`` (the same ``base_agent`` +
|
|
17
|
+
``search_space`` whole-agent contract) with ``world.kind=voice_telephony``;
|
|
18
|
+
* the voice sub-attribution is an additive tag stamped alongside the base
|
|
19
|
+
``FAILURE_LAYERS`` tag (the existing ``practice/_diagnose.py`` machinery is
|
|
20
|
+
consumed, not rewritten).
|
|
21
|
+
|
|
22
|
+
The canon constants below are this module's home; ``trinity.py`` carries literal
|
|
23
|
+
mirrors that the milestone test cross-pins (the GUNA_AXES cross-pin pattern —
|
|
24
|
+
trinity never imports this module so the gate runs even if this is broken).
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
from __future__ import annotations
|
|
28
|
+
|
|
29
|
+
from typing import Any, Mapping, Optional, Sequence
|
|
30
|
+
|
|
31
|
+
# --- canon (ARCH §4 voice-loss term refs + §2.3 sub-attribution) ------------
|
|
32
|
+
# the gate-pinned eval refs (ARCH §4 names are binding in code/gate; the UI-UX
|
|
33
|
+
# display names — tool_arg_correctness/selectivity_jir/asr_wer_delta/
|
|
34
|
+
# perturbation_robustness_delta — are presentation aliases only).
|
|
35
|
+
V1_VOICE_LOSS_TERM_REFS = (
|
|
36
|
+
"task_success",
|
|
37
|
+
"tool_argument_correctness",
|
|
38
|
+
"barge_in_latency",
|
|
39
|
+
"ttfb",
|
|
40
|
+
"wer_delta",
|
|
41
|
+
"recovery",
|
|
42
|
+
"selectivity",
|
|
43
|
+
"codec_survival",
|
|
44
|
+
"perturbation_robustness",
|
|
45
|
+
)
|
|
46
|
+
# the non-timing quality anchors — a voice loss MUST carry >= 1 of these (9A-A4).
|
|
47
|
+
V1_VOICE_LOSS_NON_TIMING_QUALITY_TERMS = ("task_success", "tool_argument_correctness")
|
|
48
|
+
# the timing-only terms (a single-timing-term objective is reward-hackable —
|
|
49
|
+
# ASPIRin R§2.3 — and is structurally rejected).
|
|
50
|
+
V1_VOICE_LOSS_TIMING_TERMS = ("barge_in_latency", "ttfb", "recovery")
|
|
51
|
+
# the four-token voice sub-attribution closed set (9A-A14), stamped alongside
|
|
52
|
+
# the base FAILURE_LAYERS tag.
|
|
53
|
+
V1_VOICE_FAILURE_SUBLAYERS = ("acoustic_codec", "asr_mishear", "llm", "tts_endpointing")
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
class VoiceLossCompositionError(ValueError):
|
|
57
|
+
"""Raised when a voice objective violates the 9A-A4 composition rule
|
|
58
|
+
(the ``voice_loss_guard_missing`` finding — a voice specialization of
|
|
59
|
+
``objective_guards_missing``)."""
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def _term_refs(objective: Mapping[str, Any]) -> list[str]:
|
|
63
|
+
return [
|
|
64
|
+
str(term.get("eval"))
|
|
65
|
+
for term in (objective.get("evals") or [])
|
|
66
|
+
if isinstance(term, Mapping) and term.get("eval")
|
|
67
|
+
]
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def compile_voice_objective(payload: Mapping[str, Any]) -> dict:
|
|
71
|
+
"""Compile a multi-objective voice loss (9A-A4). Enforces, ON TOP of the
|
|
72
|
+
verbatim ``loss.compile_objective`` Goodhart guard:
|
|
73
|
+
|
|
74
|
+
(a) >= 2 terms,
|
|
75
|
+
(b) >= 1 non-timing quality term
|
|
76
|
+
(``task_success`` / ``tool_argument_correctness``),
|
|
77
|
+
(c) a populated guard block (delegated to ``compile_objective``).
|
|
78
|
+
|
|
79
|
+
A single-timing-term voice objective is structurally rejected
|
|
80
|
+
(``VoiceLossCompositionError`` / the ``voice_loss_guard_missing`` finding).
|
|
81
|
+
The underlying guard is the UNEDITED ``loss.py`` enforcement."""
|
|
82
|
+
|
|
83
|
+
from . import loss as _loss # downward facade import (legal)
|
|
84
|
+
|
|
85
|
+
refs = _term_refs(payload)
|
|
86
|
+
if len(refs) < 2:
|
|
87
|
+
raise VoiceLossCompositionError(
|
|
88
|
+
"voice_loss_guard_missing: a voice objective is reward-hackable as a "
|
|
89
|
+
"single term (ASPIRin); it MUST be multi-objective (>= 2 terms). "
|
|
90
|
+
f"got {refs}"
|
|
91
|
+
)
|
|
92
|
+
if not any(ref in V1_VOICE_LOSS_NON_TIMING_QUALITY_TERMS for ref in refs):
|
|
93
|
+
raise VoiceLossCompositionError(
|
|
94
|
+
"voice_loss_guard_missing: a voice objective MUST carry >= 1 "
|
|
95
|
+
"non-timing quality term "
|
|
96
|
+
f"({V1_VOICE_LOSS_NON_TIMING_QUALITY_TERMS}); a timing-only loss is "
|
|
97
|
+
f"reward-hackable and is structurally rejected. got {refs}"
|
|
98
|
+
)
|
|
99
|
+
for ref in refs:
|
|
100
|
+
if ref not in V1_VOICE_LOSS_TERM_REFS:
|
|
101
|
+
raise VoiceLossCompositionError(
|
|
102
|
+
f"voice_loss_guard_missing: unknown voice loss term {ref!r}; "
|
|
103
|
+
f"expected members of {V1_VOICE_LOSS_TERM_REFS}"
|
|
104
|
+
)
|
|
105
|
+
# the verbatim Goodhart guard (loss.py:106-116) — "There is no override."
|
|
106
|
+
return _loss.compile_objective(payload)
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def attribute_voice_sublayer(
|
|
110
|
+
*,
|
|
111
|
+
failure_layer: str,
|
|
112
|
+
deficit: Mapping[str, Any] | None = None,
|
|
113
|
+
signal: str | None = None,
|
|
114
|
+
) -> str:
|
|
115
|
+
"""Map a weak voice cell to a ``V1_VOICE_FAILURE_SUBLAYERS`` token, stamped
|
|
116
|
+
ALONGSIDE the base ``FAILURE_LAYERS`` tag (a weak cell carries both). The
|
|
117
|
+
base attribution rides the existing ``practice/_diagnose.py`` machinery; this
|
|
118
|
+
is the thin sublayer helper.
|
|
119
|
+
|
|
120
|
+
Mapping (ARCH §2.3): a weak ``selectivity`` / endpointing signal →
|
|
121
|
+
``tts_endpointing`` (not ``llm``); a mis-heard value under clean audio →
|
|
122
|
+
``asr_mishear``; a claim that died through the codec → ``acoustic_codec``;
|
|
123
|
+
otherwise the reasoning/policy layer → ``llm``."""
|
|
124
|
+
|
|
125
|
+
sig = str(signal or (deficit or {}).get("signal") or "").lower()
|
|
126
|
+
if any(k in sig for k in ("selectivity", "endpoint", "barge", "vad", "interrupt", "recovery", "ttfb")):
|
|
127
|
+
return "tts_endpointing"
|
|
128
|
+
if any(k in sig for k in ("codec", "survival", "packet", "band_energy", "perturbation")):
|
|
129
|
+
return "acoustic_codec"
|
|
130
|
+
if any(k in sig for k in ("wer", "mishear", "asr", "tool_argument", "transcription")):
|
|
131
|
+
return "asr_mishear"
|
|
132
|
+
# default to the reasoning/policy layer unless infra clearly implicated.
|
|
133
|
+
if failure_layer in ("lane_infra", "framework_runtime", "provider"):
|
|
134
|
+
return "acoustic_codec"
|
|
135
|
+
return "llm"
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def build_voice_practice_loop_manifest(
|
|
139
|
+
*,
|
|
140
|
+
name: str,
|
|
141
|
+
base_agent: Mapping[str, Any],
|
|
142
|
+
search_space: Mapping[str, Sequence[Any]],
|
|
143
|
+
objective: Mapping[str, Any],
|
|
144
|
+
eval_budget: int,
|
|
145
|
+
seed: int,
|
|
146
|
+
scenario_inline: Optional[Mapping[str, Any]] = None,
|
|
147
|
+
max_rounds: int = 8,
|
|
148
|
+
) -> dict[str, Any]:
|
|
149
|
+
"""Assemble the voice improvement-loop manifest: the 13D Practice Loop on
|
|
150
|
+
``world.kind=voice_telephony`` with the multi-objective voice loss + the
|
|
151
|
+
whole voice-agent search space (9A-D5). Delegates to
|
|
152
|
+
``optimize.build_practice_loop_manifest`` so its validators hold VERBATIM.
|
|
153
|
+
The objective is compiled by ``compile_voice_objective`` (the 9A-A4 rule)
|
|
154
|
+
before it rides the simulation."""
|
|
155
|
+
|
|
156
|
+
from . import optimize as _optimize # downward facade import (legal)
|
|
157
|
+
|
|
158
|
+
compiled = compile_voice_objective(objective)
|
|
159
|
+
inline = dict(scenario_inline or {})
|
|
160
|
+
inline.setdefault("version", "agent-learning.simulation.v1")
|
|
161
|
+
inline["objective"] = compiled
|
|
162
|
+
world = dict(inline.get("world") or {})
|
|
163
|
+
world["kind"] = "voice_telephony"
|
|
164
|
+
inline["world"] = world
|
|
165
|
+
|
|
166
|
+
return _optimize.build_practice_loop_manifest(
|
|
167
|
+
name=name,
|
|
168
|
+
simulation={"version": inline["version"], "inline": inline},
|
|
169
|
+
base_agent=base_agent,
|
|
170
|
+
search_space=search_space,
|
|
171
|
+
eval_budget=eval_budget,
|
|
172
|
+
seed=seed,
|
|
173
|
+
max_rounds=max_rounds,
|
|
174
|
+
)
|
fi/api/__init__.py
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Low-level HTTP primitives used by the cloud eval clients."""
|
fi/api/auth.py
ADDED
|
@@ -0,0 +1,137 @@
|
|
|
1
|
+
"""Auth + HTTP client primitives.
|
|
2
|
+
|
|
3
|
+
``APIKeyAuth`` is the common base for every cloud client in the SDK. It
|
|
4
|
+
reads credentials from env vars (``FI_API_KEY`` / ``FI_SECRET_KEY``)
|
|
5
|
+
unless explicit keys are passed.
|
|
6
|
+
"""
|
|
7
|
+
import os
|
|
8
|
+
import time
|
|
9
|
+
from abc import ABC, abstractmethod
|
|
10
|
+
from typing import Dict, Generic, Optional, TypeVar, Union
|
|
11
|
+
|
|
12
|
+
from requests import Response
|
|
13
|
+
from requests_futures.sessions import FuturesSession
|
|
14
|
+
|
|
15
|
+
from fi.api.types import RequestConfig
|
|
16
|
+
from fi.utils.constants import (
|
|
17
|
+
API_KEY_ENVVAR_NAME,
|
|
18
|
+
DEFAULT_MAX_QUEUE,
|
|
19
|
+
DEFAULT_MAX_WORKERS,
|
|
20
|
+
DEFAULT_TIMEOUT,
|
|
21
|
+
SECRET_KEY_ENVVAR_NAME,
|
|
22
|
+
get_base_url,
|
|
23
|
+
)
|
|
24
|
+
from fi.utils.errors import DatasetNotFoundError, MissingAuthError
|
|
25
|
+
from fi.utils.executor import BoundedExecutor
|
|
26
|
+
|
|
27
|
+
T = TypeVar("T")
|
|
28
|
+
U = TypeVar("U")
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class ResponseHandler(Generic[T, U], ABC):
|
|
32
|
+
"""Parses + validates a requests.Response into a typed result."""
|
|
33
|
+
|
|
34
|
+
@classmethod
|
|
35
|
+
def parse(cls, response: Response) -> Union[T, U]:
|
|
36
|
+
if not response.ok or response.status_code != 200:
|
|
37
|
+
cls._handle_error(response)
|
|
38
|
+
return cls._parse_success(response)
|
|
39
|
+
|
|
40
|
+
@classmethod
|
|
41
|
+
@abstractmethod
|
|
42
|
+
def _parse_success(cls, response: Response) -> Union[T, U]:
|
|
43
|
+
...
|
|
44
|
+
|
|
45
|
+
@classmethod
|
|
46
|
+
@abstractmethod
|
|
47
|
+
def _handle_error(cls, response: Response) -> None:
|
|
48
|
+
...
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
class HttpClient:
|
|
52
|
+
"""Thin async-capable HTTP client with retries."""
|
|
53
|
+
|
|
54
|
+
def __init__(
|
|
55
|
+
self,
|
|
56
|
+
base_url: Optional[str] = None,
|
|
57
|
+
session: Optional[FuturesSession] = None,
|
|
58
|
+
default_headers: Optional[Dict[str, str]] = None,
|
|
59
|
+
**kwargs,
|
|
60
|
+
):
|
|
61
|
+
self._base_url = (base_url or get_base_url()).rstrip("/")
|
|
62
|
+
self._session = session or FuturesSession(
|
|
63
|
+
executor=BoundedExecutor(
|
|
64
|
+
bound=kwargs.get("max_queue", DEFAULT_MAX_QUEUE),
|
|
65
|
+
max_workers=kwargs.get("max_workers", DEFAULT_MAX_WORKERS),
|
|
66
|
+
),
|
|
67
|
+
)
|
|
68
|
+
self._default_headers = default_headers or {}
|
|
69
|
+
self._default_timeout = kwargs.get("timeout", DEFAULT_TIMEOUT)
|
|
70
|
+
|
|
71
|
+
def request(
|
|
72
|
+
self,
|
|
73
|
+
config: RequestConfig,
|
|
74
|
+
response_handler: Optional[ResponseHandler[T, U]] = None,
|
|
75
|
+
) -> Union[Response, T]:
|
|
76
|
+
url = config.url
|
|
77
|
+
headers = {**self._default_headers, **(config.headers or {})}
|
|
78
|
+
params = config.params or {}
|
|
79
|
+
json_body = config.json or {}
|
|
80
|
+
timeout = config.timeout or self._default_timeout
|
|
81
|
+
files = config.files or {}
|
|
82
|
+
data = config.data or {}
|
|
83
|
+
|
|
84
|
+
for attempt in range(config.retry_attempts):
|
|
85
|
+
try:
|
|
86
|
+
response = self._session.request(
|
|
87
|
+
method=config.method.value,
|
|
88
|
+
url=url,
|
|
89
|
+
headers=headers,
|
|
90
|
+
params=params,
|
|
91
|
+
json=json_body,
|
|
92
|
+
data=data,
|
|
93
|
+
files=files,
|
|
94
|
+
timeout=timeout,
|
|
95
|
+
).result()
|
|
96
|
+
|
|
97
|
+
if response_handler:
|
|
98
|
+
return response_handler.parse(response=response)
|
|
99
|
+
return response
|
|
100
|
+
|
|
101
|
+
except Exception as exc:
|
|
102
|
+
if isinstance(exc, DatasetNotFoundError):
|
|
103
|
+
raise
|
|
104
|
+
if attempt == config.retry_attempts - 1:
|
|
105
|
+
raise
|
|
106
|
+
time.sleep(config.retry_delay)
|
|
107
|
+
|
|
108
|
+
def close(self) -> None:
|
|
109
|
+
self._session.close()
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
class APIKeyAuth(HttpClient):
|
|
113
|
+
"""HTTP client that injects FutureAGI API + secret key headers."""
|
|
114
|
+
|
|
115
|
+
_fi_api_key: Optional[str] = None
|
|
116
|
+
_fi_secret_key: Optional[str] = None
|
|
117
|
+
|
|
118
|
+
def __init__(
|
|
119
|
+
self,
|
|
120
|
+
fi_api_key: Optional[str] = None,
|
|
121
|
+
fi_secret_key: Optional[str] = None,
|
|
122
|
+
fi_base_url: Optional[str] = None,
|
|
123
|
+
**kwargs,
|
|
124
|
+
) -> None:
|
|
125
|
+
self.__class__._fi_api_key = fi_api_key or os.environ.get(API_KEY_ENVVAR_NAME)
|
|
126
|
+
self.__class__._fi_secret_key = fi_secret_key or os.environ.get(SECRET_KEY_ENVVAR_NAME)
|
|
127
|
+
if self._fi_api_key is None or self._fi_secret_key is None:
|
|
128
|
+
raise MissingAuthError(self._fi_api_key, self._fi_secret_key)
|
|
129
|
+
|
|
130
|
+
super().__init__(
|
|
131
|
+
base_url=fi_base_url,
|
|
132
|
+
default_headers={
|
|
133
|
+
"X-Api-Key": self._fi_api_key,
|
|
134
|
+
"X-Secret-Key": self._fi_secret_key,
|
|
135
|
+
},
|
|
136
|
+
**kwargs,
|
|
137
|
+
)
|
fi/api/types.py
ADDED
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
from enum import Enum
|
|
2
|
+
from typing import Any, Dict, Optional
|
|
3
|
+
|
|
4
|
+
from pydantic import BaseModel, ConfigDict
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
class HttpMethod(Enum):
|
|
8
|
+
GET = "GET"
|
|
9
|
+
POST = "POST"
|
|
10
|
+
PUT = "PUT"
|
|
11
|
+
DELETE = "DELETE"
|
|
12
|
+
PATCH = "PATCH"
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class RequestConfig(BaseModel):
|
|
16
|
+
"""Configuration for an HTTP request."""
|
|
17
|
+
|
|
18
|
+
method: HttpMethod
|
|
19
|
+
url: str
|
|
20
|
+
headers: Optional[Dict[str, str]] = {}
|
|
21
|
+
params: Optional[Dict[str, Any]] = {}
|
|
22
|
+
files: Optional[Dict[str, Any]] = {}
|
|
23
|
+
data: Optional[Dict[str, Any]] = {}
|
|
24
|
+
json: Optional[Dict[str, Any]] = {}
|
|
25
|
+
timeout: Optional[int] = None
|
|
26
|
+
retry_attempts: int = 3
|
|
27
|
+
retry_delay: float = 1.0
|
|
28
|
+
|
|
29
|
+
model_config = ConfigDict(protected_namespaces=())
|
fi/cli/__init__.py
ADDED
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
"""Assertions module for CLI evaluation result validation."""
|
|
2
|
+
|
|
3
|
+
from .conditions import Condition, Operator, MetricType
|
|
4
|
+
from .parser import ConditionParser
|
|
5
|
+
from .evaluator import (
|
|
6
|
+
AssertionEvaluator,
|
|
7
|
+
AssertionResult,
|
|
8
|
+
AssertionOutcome,
|
|
9
|
+
AssertionReport,
|
|
10
|
+
)
|
|
11
|
+
from .reporter import AssertionReporter
|
|
12
|
+
from .exit_codes import ExitCode
|
|
13
|
+
|
|
14
|
+
__all__ = [
|
|
15
|
+
"Condition",
|
|
16
|
+
"Operator",
|
|
17
|
+
"MetricType",
|
|
18
|
+
"ConditionParser",
|
|
19
|
+
"AssertionEvaluator",
|
|
20
|
+
"AssertionResult",
|
|
21
|
+
"AssertionOutcome",
|
|
22
|
+
"AssertionReport",
|
|
23
|
+
"AssertionReporter",
|
|
24
|
+
"ExitCode",
|
|
25
|
+
]
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
"""Condition types and operators for assertions."""
|
|
2
|
+
|
|
3
|
+
from enum import Enum
|
|
4
|
+
from typing import Union, Optional
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class Operator(Enum):
|
|
9
|
+
"""Comparison operators for assertion conditions."""
|
|
10
|
+
GTE = ">=" # Greater than or equal
|
|
11
|
+
LTE = "<=" # Less than or equal
|
|
12
|
+
GT = ">" # Greater than
|
|
13
|
+
LT = "<" # Less than
|
|
14
|
+
EQ = "==" # Equal
|
|
15
|
+
NEQ = "!=" # Not equal
|
|
16
|
+
BETWEEN = "between" # Between two values
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class MetricType(Enum):
|
|
20
|
+
"""Types of metrics that can be evaluated in assertions."""
|
|
21
|
+
PASS_RATE = "pass_rate" # Percentage of passing evaluations
|
|
22
|
+
AVG_SCORE = "avg_score" # Average score (for numeric outputs)
|
|
23
|
+
MIN_SCORE = "min_score" # Minimum score
|
|
24
|
+
MAX_SCORE = "max_score" # Maximum score
|
|
25
|
+
FAILED_COUNT = "failed_count" # Number of failures
|
|
26
|
+
PASSED_COUNT = "passed_count" # Number of passes
|
|
27
|
+
TOTAL_COUNT = "total_count" # Total evaluations
|
|
28
|
+
P50_SCORE = "p50_score" # 50th percentile
|
|
29
|
+
P90_SCORE = "p90_score" # 90th percentile
|
|
30
|
+
P95_SCORE = "p95_score" # 95th percentile
|
|
31
|
+
RUNTIME_AVG = "runtime_avg" # Average runtime in ms
|
|
32
|
+
RUNTIME_P95 = "runtime_p95" # 95th percentile runtime
|
|
33
|
+
# Global metrics (across all templates)
|
|
34
|
+
TOTAL_PASS_RATE = "total_pass_rate"
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
@dataclass
|
|
38
|
+
class Condition:
|
|
39
|
+
"""A single assertion condition."""
|
|
40
|
+
metric: MetricType
|
|
41
|
+
operator: Operator
|
|
42
|
+
value: Union[float, int]
|
|
43
|
+
value2: Optional[Union[float, int]] = None # For BETWEEN operator
|
|
44
|
+
|
|
45
|
+
def evaluate(self, actual_value: float) -> bool:
|
|
46
|
+
"""Evaluate if the condition passes.
|
|
47
|
+
|
|
48
|
+
Args:
|
|
49
|
+
actual_value: The actual metric value to compare against.
|
|
50
|
+
|
|
51
|
+
Returns:
|
|
52
|
+
True if the condition passes, False otherwise.
|
|
53
|
+
"""
|
|
54
|
+
if self.operator == Operator.GTE:
|
|
55
|
+
return actual_value >= self.value
|
|
56
|
+
elif self.operator == Operator.LTE:
|
|
57
|
+
return actual_value <= self.value
|
|
58
|
+
elif self.operator == Operator.GT:
|
|
59
|
+
return actual_value > self.value
|
|
60
|
+
elif self.operator == Operator.LT:
|
|
61
|
+
return actual_value < self.value
|
|
62
|
+
elif self.operator == Operator.EQ:
|
|
63
|
+
return actual_value == self.value
|
|
64
|
+
elif self.operator == Operator.NEQ:
|
|
65
|
+
return actual_value != self.value
|
|
66
|
+
elif self.operator == Operator.BETWEEN:
|
|
67
|
+
if self.value2 is None:
|
|
68
|
+
return False
|
|
69
|
+
return self.value <= actual_value <= self.value2
|
|
70
|
+
return False
|
|
71
|
+
|
|
72
|
+
def __str__(self) -> str:
|
|
73
|
+
"""String representation of the condition."""
|
|
74
|
+
if self.operator == Operator.BETWEEN:
|
|
75
|
+
return f"{self.value} <= {self.metric.value} <= {self.value2}"
|
|
76
|
+
return f"{self.metric.value} {self.operator.value} {self.value}"
|