agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,376 @@
|
|
|
1
|
+
"""LiveKit live lane (3B) — real ``livekit-agents`` AgentSession, opt-in.
|
|
2
|
+
|
|
3
|
+
Framework imports: NONE at module top (P3-D1). Rung-1 execution happens in
|
|
4
|
+
the ``_workers/livekit_worker.py`` subprocess (the only sanctioned top-level
|
|
5
|
+
framework import home); this module is importable in the no-extras release
|
|
6
|
+
env and the live_lane_boundary gate scans it like any release module.
|
|
7
|
+
|
|
8
|
+
Rungs (P3-D3): 1 virtual-clock text driver (default, implemented) →
|
|
9
|
+
2 loopback real-transport audio → 3 LiveKit Cloud/SIP (``live_credentialed``,
|
|
10
|
+
standard LiveKit credential names). Rung 1 is honest about its tier: timing-only voice metrics,
|
|
11
|
+
no ``channels`` block, no audio claims (guide §3.5).
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import tempfile
|
|
17
|
+
import uuid
|
|
18
|
+
from pathlib import Path
|
|
19
|
+
from typing import Any, Mapping, Optional, Sequence
|
|
20
|
+
|
|
21
|
+
from ._contract import lane_budget_s, require_lane_enabled
|
|
22
|
+
from ._perturb import apply_text_perturbations, perturbations_stanza
|
|
23
|
+
from ._runner import run_worker_once
|
|
24
|
+
from ._stats import (
|
|
25
|
+
derive_channel_evidence,
|
|
26
|
+
lane_run_payload,
|
|
27
|
+
primary_transcript_events,
|
|
28
|
+
run_repeated,
|
|
29
|
+
)
|
|
30
|
+
|
|
31
|
+
_WORKERS = Path(__file__).resolve().parent / "_workers"
|
|
32
|
+
_RUNG_LABELS = {1: "virtual_clock", 2: "loopback_transport", 3: "cloud_sip"}
|
|
33
|
+
|
|
34
|
+
# Rung-3 credential names: exactly the names the vendored engine reads
|
|
35
|
+
# (engines/livekit.py reads LIVEKIT_API_KEY/LIVEKIT_API_SECRET; the server
|
|
36
|
+
# URL arrives via LIVEKIT_URL, P3-D5).
|
|
37
|
+
RUNG3_REQUIRED_ENV = ("LIVEKIT_URL", "LIVEKIT_API_KEY", "LIVEKIT_API_SECRET")
|
|
38
|
+
|
|
39
|
+
_DEFAULT_TURNS = (
|
|
40
|
+
{"user": "Hello, can you hear me?"},
|
|
41
|
+
{"user": "Great - please confirm my appointment for tomorrow."},
|
|
42
|
+
)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _scenario_turns(scenario: Mapping[str, Any]) -> list[dict[str, Any]]:
|
|
46
|
+
raw = scenario.get("turns") or scenario.get("user_messages")
|
|
47
|
+
if not raw:
|
|
48
|
+
return [dict(turn) for turn in _DEFAULT_TURNS]
|
|
49
|
+
turns: list[dict[str, Any]] = []
|
|
50
|
+
for item in raw:
|
|
51
|
+
if isinstance(item, str):
|
|
52
|
+
turns.append({"user": item})
|
|
53
|
+
elif isinstance(item, Mapping):
|
|
54
|
+
turns.append(dict(item))
|
|
55
|
+
return turns or [dict(turn) for turn in _DEFAULT_TURNS]
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def _voice_timing(events: Sequence[Mapping[str, Any]]) -> dict[str, Any]:
|
|
59
|
+
"""Timing-only voice metrics (the rung-1 honesty tier): per-turn agent
|
|
60
|
+
response latency derived from event timestamps — no audio claims."""
|
|
61
|
+
|
|
62
|
+
latencies_ms: list[float] = []
|
|
63
|
+
pending_user_t: float | None = None
|
|
64
|
+
for event in events:
|
|
65
|
+
channel = event.get("channel")
|
|
66
|
+
if channel == "user" and event.get("type") == "message":
|
|
67
|
+
t = event.get("t")
|
|
68
|
+
pending_user_t = float(t) if isinstance(t, (int, float)) else None
|
|
69
|
+
elif channel == "agent" and event.get("type") == "message":
|
|
70
|
+
t = event.get("t")
|
|
71
|
+
if pending_user_t is not None and isinstance(t, (int, float)):
|
|
72
|
+
latencies_ms.append(round((float(t) - pending_user_t) * 1000.0, 3))
|
|
73
|
+
pending_user_t = None
|
|
74
|
+
return {
|
|
75
|
+
"turn_latencies_ms": latencies_ms,
|
|
76
|
+
"mean_turn_latency_ms": (
|
|
77
|
+
round(sum(latencies_ms) / len(latencies_ms), 3) if latencies_ms else None
|
|
78
|
+
),
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def _realtime_state(
|
|
83
|
+
events: Sequence[Mapping[str, Any]], *, rung_label: str
|
|
84
|
+
) -> dict[str, Any]:
|
|
85
|
+
items = []
|
|
86
|
+
for index, event in enumerate(events, start=1):
|
|
87
|
+
if event.get("channel") in ("user", "agent", "tool"):
|
|
88
|
+
payload = event.get("payload")
|
|
89
|
+
payload = payload if isinstance(payload, Mapping) else {}
|
|
90
|
+
items.append(
|
|
91
|
+
{
|
|
92
|
+
"index": index,
|
|
93
|
+
"channel": event.get("channel"),
|
|
94
|
+
"item_type": event.get("type"),
|
|
95
|
+
"text": payload.get("text"),
|
|
96
|
+
}
|
|
97
|
+
)
|
|
98
|
+
return {
|
|
99
|
+
"engine": "live_lane_livekit",
|
|
100
|
+
"rung": rung_label,
|
|
101
|
+
"item_count": len(items),
|
|
102
|
+
"items": items[:200],
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def _rung2_loopback_channels(
|
|
107
|
+
turns: Sequence[Mapping[str, Any]],
|
|
108
|
+
*,
|
|
109
|
+
loopback: Optional[Mapping[str, Any]],
|
|
110
|
+
codec_profile: str,
|
|
111
|
+
seed: int,
|
|
112
|
+
acoustic_operators: Sequence[str] = (),
|
|
113
|
+
) -> tuple[dict[str, Any], str, list[dict[str, Any]]]:
|
|
114
|
+
"""Phase 9A unit 2 + Phase-12 12C rung-2 — the rung-2 loopback dispatch
|
|
115
|
+
(§2.1 / §2.5 + ARCH §2c).
|
|
116
|
+
|
|
117
|
+
Produce the two PCM streams via the deterministic ``_loopback`` round-trip,
|
|
118
|
+
apply the rung-2 ACOUSTIC operators (Phase-12 12C: ``mix_noise`` /
|
|
119
|
+
``mix_interference`` / ``reverb_blend`` over the user PCM — the attack the
|
|
120
|
+
framework hears) BEFORE the codec stage, apply the default-ON codec
|
|
121
|
+
round-trip (9A-A11) unless ``codec_profile == "none"``, feed the
|
|
122
|
+
ALREADY-BUILT ``derive_channel_evidence`` (REUSED, NOT rebuilt), and return
|
|
123
|
+
the ``channels`` block + the ``fidelity_tier`` marker + the applied acoustic
|
|
124
|
+
operator records (the paired-clean stanza). The loopback module is reached
|
|
125
|
+
via the sanctioned ``from fi.alk import live`` function-body idiom so
|
|
126
|
+
this module stays framework-free and the ``live_lane_boundary`` import
|
|
127
|
+
discipline holds.
|
|
128
|
+
|
|
129
|
+
The codec-survival score is computed on the PERTURBED-then-channel signal so
|
|
130
|
+
``phone_survival`` honestly reflects whether the acoustic attack reproduces
|
|
131
|
+
through the 8 kHz telephony channel (P12-D2): no ``survives``/``partial``
|
|
132
|
+
claim without a codec record."""
|
|
133
|
+
|
|
134
|
+
from fi.alk import live # sanctioned facade idiom (cli.py)
|
|
135
|
+
|
|
136
|
+
cfg = dict(loopback or {})
|
|
137
|
+
tick_ms = float(cfg.get("tick_ms", live._loopback.DEFAULT_TICK_MS))
|
|
138
|
+
sample_rate = int(cfg.get("sample_rate", live._loopback.DEFAULT_SAMPLE_RATE))
|
|
139
|
+
loop_seed = int(cfg.get("seed", seed))
|
|
140
|
+
profile = str(cfg.get("codec_profile", codec_profile))
|
|
141
|
+
|
|
142
|
+
loop = live._loopback.run_loopback_roundtrip(
|
|
143
|
+
list(turns),
|
|
144
|
+
user_wav=cfg.get("user_wav"),
|
|
145
|
+
agent_wav=cfg.get("agent_wav"),
|
|
146
|
+
tick_ms=tick_ms,
|
|
147
|
+
sample_rate=sample_rate,
|
|
148
|
+
seed=loop_seed,
|
|
149
|
+
)
|
|
150
|
+
user_pcm, agent_pcm = loop["user_pcm"], loop["agent_pcm"]
|
|
151
|
+
|
|
152
|
+
# Phase-12 12C rung-2: the acoustic attack rides the USER channel (the side
|
|
153
|
+
# the framework hears). Applied to the CLEAN loopback PCM before the codec
|
|
154
|
+
# stage; deterministic under loop_seed. The agent side is untouched.
|
|
155
|
+
acoustic_applied: list[dict[str, Any]] = []
|
|
156
|
+
attacked_user_pcm = user_pcm
|
|
157
|
+
if acoustic_operators:
|
|
158
|
+
attacked_user_pcm, acoustic_applied = live._perturb.apply_acoustic_perturbations(
|
|
159
|
+
user_pcm,
|
|
160
|
+
list(acoustic_operators),
|
|
161
|
+
seed=loop_seed,
|
|
162
|
+
sample_rate=sample_rate,
|
|
163
|
+
)
|
|
164
|
+
user_pcm = attacked_user_pcm
|
|
165
|
+
|
|
166
|
+
codec_record: dict[str, Any] | None = None
|
|
167
|
+
phone_survival: dict[str, Any] | None = None
|
|
168
|
+
if profile != "none":
|
|
169
|
+
user_pcm, agent_pcm, codec_record = live._codec.apply_codec_profile(
|
|
170
|
+
user_pcm, agent_pcm, profile=profile, seed=loop_seed, sample_rate=sample_rate
|
|
171
|
+
)
|
|
172
|
+
codec, packet_loss = live._codec._PROFILE_BUNDLE[profile]
|
|
173
|
+
# the attack rides the USER channel, so re-validate the user side through
|
|
174
|
+
# the channel (the clean user PCM is the pre-channel twin).
|
|
175
|
+
phone_survival = live._codec.score_codec_survival(
|
|
176
|
+
loop["user_pcm"],
|
|
177
|
+
attacked_user_pcm,
|
|
178
|
+
codec=codec,
|
|
179
|
+
packet_loss=packet_loss,
|
|
180
|
+
seed=loop_seed,
|
|
181
|
+
sample_rate=sample_rate,
|
|
182
|
+
)
|
|
183
|
+
|
|
184
|
+
derived = derive_channel_evidence(
|
|
185
|
+
user_pcm, agent_pcm, sample_rate=(8000 if profile != "none" else sample_rate)
|
|
186
|
+
)
|
|
187
|
+
channels: dict[str, Any] = {
|
|
188
|
+
"derived": derived,
|
|
189
|
+
"source": "derive_channel_evidence",
|
|
190
|
+
"rung": _RUNG_LABELS[2],
|
|
191
|
+
"fidelity_tier": "deterministic_loopback",
|
|
192
|
+
"seed": loop_seed,
|
|
193
|
+
"loopback_provenance": loop["provenance"],
|
|
194
|
+
}
|
|
195
|
+
if codec_record is not None:
|
|
196
|
+
channels["codec_round_trip"] = codec_record
|
|
197
|
+
if phone_survival is not None:
|
|
198
|
+
channels["phone_survival"] = phone_survival
|
|
199
|
+
if acoustic_applied:
|
|
200
|
+
channels["acoustic_operators"] = acoustic_applied
|
|
201
|
+
return channels, "deterministic_loopback", acoustic_applied
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def run_livekit_lane(
|
|
205
|
+
scenario: Mapping[str, Any],
|
|
206
|
+
*,
|
|
207
|
+
rung: int = 1, # P3-D3: 1 virtual-clock | 2 loopback transport | 3 cloud/SIP
|
|
208
|
+
repeats: int = 8,
|
|
209
|
+
stressed: bool = False, # perturbation sub-lane -> evidence_class "live_stressed"
|
|
210
|
+
perturbations: Optional[Sequence[str]] = None,
|
|
211
|
+
seed: int = 0,
|
|
212
|
+
required_env: Optional[Sequence[str]] = None,
|
|
213
|
+
version_requirement: str | None = None,
|
|
214
|
+
budget_s: float | None = None,
|
|
215
|
+
artifacts_dir: str | Path | None = None,
|
|
216
|
+
# Phase 9A (BBG A2): additive optional loopback config consumed ONLY on the
|
|
217
|
+
# rung==2 branch; rung-1/rung-3 callers are unaffected.
|
|
218
|
+
loopback: Optional[Mapping[str, Any]] = None,
|
|
219
|
+
codec_profile: str = "g711_ulaw_8k_ge",
|
|
220
|
+
) -> dict[str, Any]:
|
|
221
|
+
require_lane_enabled("livekit")
|
|
222
|
+
if rung >= 3:
|
|
223
|
+
require_lane_enabled("credentialed")
|
|
224
|
+
if rung not in _RUNG_LABELS:
|
|
225
|
+
raise ValueError(f"rung must be one of {sorted(_RUNG_LABELS)}, got {rung}")
|
|
226
|
+
|
|
227
|
+
required = tuple(required_env) if required_env is not None else ()
|
|
228
|
+
operators = list(perturbations or (["asr_error"] if stressed else []))
|
|
229
|
+
turns = _scenario_turns(scenario)
|
|
230
|
+
# Phase-12 12C rung-2: split text-rung operators (applied to the turn script)
|
|
231
|
+
# from acoustic operators (applied to the rung-2 loopback PCM). At rung-1 an
|
|
232
|
+
# acoustic operator still raises inside ``apply_text_perturbations`` (the
|
|
233
|
+
# rung wall is unchanged for text-rung input).
|
|
234
|
+
from ._perturb import ACOUSTIC_RUNG_OPERATORS
|
|
235
|
+
|
|
236
|
+
acoustic_operators = [op for op in operators if op in ACOUSTIC_RUNG_OPERATORS]
|
|
237
|
+
text_operators = [op for op in operators if op not in ACOUSTIC_RUNG_OPERATORS]
|
|
238
|
+
if rung != 2 and acoustic_operators:
|
|
239
|
+
# acoustic operators require the rung-2 PCM channel; outside it they hit
|
|
240
|
+
# the same rung wall ``apply_text_perturbations`` enforces (no silent
|
|
241
|
+
# acoustic claim before the audio channel exists — ARCH §2c).
|
|
242
|
+
raise ValueError(
|
|
243
|
+
f"acoustic operators {acoustic_operators} need a real audio channel "
|
|
244
|
+
"(rung 2 loopback transport or above); rung "
|
|
245
|
+
f"{rung} ({_RUNG_LABELS[rung]}) is a text-rung tier"
|
|
246
|
+
)
|
|
247
|
+
applied: list[dict[str, Any]] = []
|
|
248
|
+
if text_operators:
|
|
249
|
+
turns, applied = apply_text_perturbations(turns, text_operators, seed=seed)
|
|
250
|
+
|
|
251
|
+
# Phase 9A unit 2: the rung wall narrows — rung-2 dispatches into the
|
|
252
|
+
# deterministic loopback (§2.1); rung-3 still raises (the owner live-proof,
|
|
253
|
+
# unit 7). rung-1 is completely untouched (timing-only, NO channels block).
|
|
254
|
+
channels: dict[str, Any] | None = None
|
|
255
|
+
fidelity_tier: str | None = None
|
|
256
|
+
acoustic_applied: list[dict[str, Any]] = []
|
|
257
|
+
if rung == 2:
|
|
258
|
+
channels, fidelity_tier, acoustic_applied = _rung2_loopback_channels(
|
|
259
|
+
turns,
|
|
260
|
+
loopback=loopback,
|
|
261
|
+
codec_profile=codec_profile,
|
|
262
|
+
seed=seed,
|
|
263
|
+
acoustic_operators=acoustic_operators,
|
|
264
|
+
)
|
|
265
|
+
# §2.5 binding correction: a deterministic in-process loopback is
|
|
266
|
+
# NEVER live_lane. Default codec round-trip is ON (9A-A11) → a stressed
|
|
267
|
+
# run → live_stressed; a no-op (codec_profile="none") clean run is also
|
|
268
|
+
# live_stressed at rung-2 (it never claims live_lane). captured_fixture
|
|
269
|
+
# is reached through the capture flow, not here.
|
|
270
|
+
evidence_class = "live_stressed"
|
|
271
|
+
elif rung != 1:
|
|
272
|
+
# rung == 3: unchanged keyed path; still requires the credentialed flag
|
|
273
|
+
# + RUNG3_REQUIRED_ENV; rung-3 lands as the owner live-proof (unit 7).
|
|
274
|
+
raise NotImplementedError(
|
|
275
|
+
f"livekit lane rung {rung} ({_RUNG_LABELS[rung]}) is not "
|
|
276
|
+
"implemented yet; rung 1 (virtual_clock) and rung 2 "
|
|
277
|
+
"(loopback_transport) are the supported tiers — rung 3 (cloud_sip) "
|
|
278
|
+
"is the owner-keyed live-proof lane"
|
|
279
|
+
)
|
|
280
|
+
else:
|
|
281
|
+
evidence_class = "live_stressed" if operators else "live_lane"
|
|
282
|
+
|
|
283
|
+
base_dir = (
|
|
284
|
+
Path(artifacts_dir)
|
|
285
|
+
if artifacts_dir is not None
|
|
286
|
+
else Path(tempfile.mkdtemp(prefix="agent-learning-live-livekit-"))
|
|
287
|
+
)
|
|
288
|
+
run_id = uuid.uuid4().hex
|
|
289
|
+
resolved_budget = float(budget_s) if budget_s is not None else lane_budget_s("livekit")
|
|
290
|
+
boot = {
|
|
291
|
+
"type": "boot",
|
|
292
|
+
"lane": "livekit",
|
|
293
|
+
"rung": rung,
|
|
294
|
+
"scenario": {"name": str(scenario.get("name") or "livekit-smoke")},
|
|
295
|
+
"turns": turns,
|
|
296
|
+
"config": {
|
|
297
|
+
"instructions": scenario.get("instructions")
|
|
298
|
+
or "You are a concise, helpful voice agent under test.",
|
|
299
|
+
"responses": scenario.get("responses"),
|
|
300
|
+
"expect": scenario.get("expect"),
|
|
301
|
+
},
|
|
302
|
+
}
|
|
303
|
+
worker = _WORKERS / "livekit_worker.py"
|
|
304
|
+
|
|
305
|
+
def _run_once(index: int, transcript: Any) -> dict[str, Any]:
|
|
306
|
+
return run_worker_once(
|
|
307
|
+
worker,
|
|
308
|
+
boot,
|
|
309
|
+
lane="livekit",
|
|
310
|
+
required_env=required,
|
|
311
|
+
cwd=base_dir,
|
|
312
|
+
timeout_s=resolved_budget,
|
|
313
|
+
transcript=transcript,
|
|
314
|
+
version_requirement=version_requirement,
|
|
315
|
+
)
|
|
316
|
+
|
|
317
|
+
result = run_repeated(
|
|
318
|
+
_run_once,
|
|
319
|
+
lane="livekit",
|
|
320
|
+
evidence_class=evidence_class,
|
|
321
|
+
repeats=repeats,
|
|
322
|
+
budget_s=budget_s,
|
|
323
|
+
required_env=required,
|
|
324
|
+
artifacts_dir=base_dir,
|
|
325
|
+
run_id=run_id,
|
|
326
|
+
rung=_RUNG_LABELS[rung],
|
|
327
|
+
framework="livekit-agents",
|
|
328
|
+
version_requirement=version_requirement,
|
|
329
|
+
)
|
|
330
|
+
|
|
331
|
+
events = primary_transcript_events(result)
|
|
332
|
+
# Normalization rides the existing realtime manifest builder — the run
|
|
333
|
+
# lands in the existing `realtime_trace` state family; the live engine
|
|
334
|
+
# is declared in metadata (guide §3.1).
|
|
335
|
+
from .. import simulate as _simulate
|
|
336
|
+
|
|
337
|
+
manifest = _simulate.build_realtime_run_manifest(
|
|
338
|
+
name=f"live-livekit-{run_id[:8]}",
|
|
339
|
+
framework="livekit",
|
|
340
|
+
required_env=required,
|
|
341
|
+
min_turns=1,
|
|
342
|
+
max_turns=max(len(turns), 1),
|
|
343
|
+
metadata={
|
|
344
|
+
"simulation_engine": "live_lane_livekit",
|
|
345
|
+
"live_lane": {"lane": "livekit", "rung": _RUNG_LABELS[rung]},
|
|
346
|
+
},
|
|
347
|
+
)
|
|
348
|
+
|
|
349
|
+
payload = lane_run_payload(
|
|
350
|
+
result,
|
|
351
|
+
name=f"live-livekit-{run_id[:8]}",
|
|
352
|
+
scenario=scenario,
|
|
353
|
+
manifest=manifest,
|
|
354
|
+
states={"realtime_trace": _realtime_state(events, rung_label=_RUNG_LABELS[rung])},
|
|
355
|
+
metadata={
|
|
356
|
+
"execution_model": "subprocess",
|
|
357
|
+
"rung": _RUNG_LABELS[rung],
|
|
358
|
+
# rung-1 honesty: timing-only voice metrics, NO channels block
|
|
359
|
+
"voice_timing": _voice_timing(events),
|
|
360
|
+
},
|
|
361
|
+
)
|
|
362
|
+
# the perturbations stanza carries BOTH families (text-rung records + the
|
|
363
|
+
# rung-2 acoustic records); the clean-twin link is filled by the campaign.
|
|
364
|
+
all_applied = list(applied) + list(acoustic_applied)
|
|
365
|
+
if all_applied:
|
|
366
|
+
payload["live_lane"]["perturbations"] = perturbations_stanza(
|
|
367
|
+
all_applied, seed=seed, paired_clean_run=None
|
|
368
|
+
)
|
|
369
|
+
if channels is not None:
|
|
370
|
+
# rung-2: attach the dual-channel evidence + the fidelity marker (§2.5 /
|
|
371
|
+
# 9A-A10). fidelity_tier is a MARKER FIELD, not a new evidence class.
|
|
372
|
+
payload["channels"] = channels
|
|
373
|
+
if isinstance(payload.get("live_lane"), dict):
|
|
374
|
+
payload["live_lane"]["fidelity_tier"] = fidelity_tier
|
|
375
|
+
payload["fidelity_tier"] = fidelity_tier
|
|
376
|
+
return payload
|
fi/alk/live/mcp_lane.py
ADDED
|
@@ -0,0 +1,172 @@
|
|
|
1
|
+
"""MCP live lane (3E) — real MCP server processes over the real protocol.
|
|
2
|
+
|
|
3
|
+
Framework imports: NONE at module top (P3-D1). The default tier spawns the
|
|
4
|
+
shipped loopback stdio server (``_workers/mcp_loopback_server.py`` — a real
|
|
5
|
+
``FastMCP`` process, credential-free but genuinely separate and speaking the
|
|
6
|
+
real protocol over the wire: that IS the live graduation, P3-D6/R§1 #12).
|
|
7
|
+
The client side is ``_workers/mcp_worker.py`` (a ``ClientSession`` over
|
|
8
|
+
stdio). Every artifact carries the server-behavior snapshot stamp
|
|
9
|
+
``{server_name, server_version, capability_hash}`` (R§1 #11).
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import sys
|
|
15
|
+
import tempfile
|
|
16
|
+
import uuid
|
|
17
|
+
from pathlib import Path
|
|
18
|
+
from typing import Any, Mapping, Optional, Sequence
|
|
19
|
+
|
|
20
|
+
from ._contract import lane_budget_s, require_lane_enabled
|
|
21
|
+
from ._runner import run_worker_once
|
|
22
|
+
from ._stats import lane_run_payload, primary_transcript_events, run_repeated
|
|
23
|
+
|
|
24
|
+
_WORKERS = Path(__file__).resolve().parent / "_workers"
|
|
25
|
+
_RUNG_LABELS = {1: "loopback_servers", 2: "third_party_servers"}
|
|
26
|
+
|
|
27
|
+
# Deterministic, credential-free default tool script against the loopback
|
|
28
|
+
# server (claim-level expectations, tolerant of alternative trajectories).
|
|
29
|
+
_DEFAULT_CALLS = (
|
|
30
|
+
{"tool": "echo", "arguments": {"text": "hello loopback"}, "expect": {"contains": "hello loopback"}},
|
|
31
|
+
{"tool": "add", "arguments": {"a": 2, "b": 3}, "expect": {"contains": "5"}},
|
|
32
|
+
)
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _scenario_calls(scenario: Mapping[str, Any]) -> list[dict[str, Any]]:
|
|
36
|
+
raw = scenario.get("calls")
|
|
37
|
+
if not raw:
|
|
38
|
+
return [dict(call) for call in _DEFAULT_CALLS]
|
|
39
|
+
return [dict(call) for call in raw if isinstance(call, Mapping)]
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _server_snapshot(
|
|
43
|
+
events: Sequence[Mapping[str, Any]],
|
|
44
|
+
) -> dict[str, Any] | None:
|
|
45
|
+
for event in events:
|
|
46
|
+
if event.get("type") == "server_snapshot":
|
|
47
|
+
payload = event.get("payload")
|
|
48
|
+
if isinstance(payload, Mapping):
|
|
49
|
+
return dict(payload)
|
|
50
|
+
return None
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def _tool_session_state(events: Sequence[Mapping[str, Any]]) -> dict[str, Any]:
|
|
54
|
+
items = []
|
|
55
|
+
for index, event in enumerate(events, start=1):
|
|
56
|
+
if event.get("channel") == "tool":
|
|
57
|
+
payload = event.get("payload")
|
|
58
|
+
payload = payload if isinstance(payload, Mapping) else {}
|
|
59
|
+
items.append(
|
|
60
|
+
{
|
|
61
|
+
"index": index,
|
|
62
|
+
"item_type": event.get("type"),
|
|
63
|
+
"tool": payload.get("name"),
|
|
64
|
+
"ok": payload.get("ok"),
|
|
65
|
+
}
|
|
66
|
+
)
|
|
67
|
+
return {
|
|
68
|
+
"engine": "live_lane_mcp",
|
|
69
|
+
"item_count": len(items),
|
|
70
|
+
"items": items[:200],
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def run_mcp_lane(
|
|
75
|
+
scenario: Mapping[str, Any],
|
|
76
|
+
*,
|
|
77
|
+
server: Optional[Mapping[str, Any]] = None,
|
|
78
|
+
repeats: int = 8,
|
|
79
|
+
required_env: Optional[Sequence[str]] = None,
|
|
80
|
+
version_requirement: str | None = None,
|
|
81
|
+
budget_s: float | None = None,
|
|
82
|
+
artifacts_dir: str | Path | None = None,
|
|
83
|
+
) -> dict[str, Any]:
|
|
84
|
+
"""Default tier (server=None): loopback stdio server fixture + client.
|
|
85
|
+
Third-party tier (server={"command": [...], "env_names": [...]}) is
|
|
86
|
+
``live_credentialed`` with server-specific names (P3-D6)."""
|
|
87
|
+
|
|
88
|
+
require_lane_enabled("mcp")
|
|
89
|
+
rung = 1 if server is None else 2
|
|
90
|
+
if rung >= 2:
|
|
91
|
+
require_lane_enabled("credentialed")
|
|
92
|
+
|
|
93
|
+
if server is None:
|
|
94
|
+
server_command = [sys.executable, str(_WORKERS / "mcp_loopback_server.py")]
|
|
95
|
+
server_env_names: list[str] = []
|
|
96
|
+
else:
|
|
97
|
+
command = server.get("command")
|
|
98
|
+
if not isinstance(command, Sequence) or not command:
|
|
99
|
+
raise ValueError(
|
|
100
|
+
"third-party server spec needs a non-empty 'command' list"
|
|
101
|
+
)
|
|
102
|
+
server_command = [str(part) for part in command]
|
|
103
|
+
server_env_names = [str(name) for name in server.get("env_names") or []]
|
|
104
|
+
required = tuple(
|
|
105
|
+
required_env if required_env is not None else server_env_names
|
|
106
|
+
)
|
|
107
|
+
|
|
108
|
+
base_dir = (
|
|
109
|
+
Path(artifacts_dir)
|
|
110
|
+
if artifacts_dir is not None
|
|
111
|
+
else Path(tempfile.mkdtemp(prefix="agent-learning-live-mcp-"))
|
|
112
|
+
)
|
|
113
|
+
run_id = uuid.uuid4().hex
|
|
114
|
+
resolved_budget = float(budget_s) if budget_s is not None else lane_budget_s("mcp")
|
|
115
|
+
boot = {
|
|
116
|
+
"type": "boot",
|
|
117
|
+
"lane": "mcp",
|
|
118
|
+
"rung": rung,
|
|
119
|
+
"scenario": {"name": str(scenario.get("name") or "mcp-loopback-smoke")},
|
|
120
|
+
"config": {
|
|
121
|
+
"server_command": server_command,
|
|
122
|
+
"server_env_names": server_env_names,
|
|
123
|
+
"calls": _scenario_calls(scenario),
|
|
124
|
+
},
|
|
125
|
+
}
|
|
126
|
+
worker = _WORKERS / "mcp_worker.py"
|
|
127
|
+
|
|
128
|
+
def _run_once(index: int, transcript: Any) -> dict[str, Any]:
|
|
129
|
+
return run_worker_once(
|
|
130
|
+
worker,
|
|
131
|
+
boot,
|
|
132
|
+
lane="mcp",
|
|
133
|
+
required_env=required,
|
|
134
|
+
cwd=base_dir,
|
|
135
|
+
timeout_s=resolved_budget,
|
|
136
|
+
transcript=transcript,
|
|
137
|
+
version_requirement=version_requirement,
|
|
138
|
+
)
|
|
139
|
+
|
|
140
|
+
result = run_repeated(
|
|
141
|
+
_run_once,
|
|
142
|
+
lane="mcp",
|
|
143
|
+
evidence_class="live_lane",
|
|
144
|
+
repeats=repeats,
|
|
145
|
+
budget_s=budget_s,
|
|
146
|
+
required_env=required,
|
|
147
|
+
artifacts_dir=base_dir,
|
|
148
|
+
run_id=run_id,
|
|
149
|
+
rung=_RUNG_LABELS[rung],
|
|
150
|
+
framework="mcp",
|
|
151
|
+
version_requirement=version_requirement,
|
|
152
|
+
)
|
|
153
|
+
|
|
154
|
+
events = primary_transcript_events(result)
|
|
155
|
+
payload = lane_run_payload(
|
|
156
|
+
result,
|
|
157
|
+
name=f"live-mcp-{run_id[:8]}",
|
|
158
|
+
scenario=scenario,
|
|
159
|
+
states={
|
|
160
|
+
"framework_runtime": {
|
|
161
|
+
"framework": "mcp",
|
|
162
|
+
"engine": "live_lane_mcp",
|
|
163
|
+
"rung": _RUNG_LABELS[rung],
|
|
164
|
+
},
|
|
165
|
+
"mcp_tool_session": _tool_session_state(events),
|
|
166
|
+
},
|
|
167
|
+
metadata={"execution_model": "subprocess", "rung": _RUNG_LABELS[rung]},
|
|
168
|
+
)
|
|
169
|
+
snapshot = _server_snapshot(events)
|
|
170
|
+
if snapshot is not None:
|
|
171
|
+
payload["live_lane"]["server_snapshot"] = snapshot
|
|
172
|
+
return payload
|