agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
fi/alk/live/_codec.py
ADDED
|
@@ -0,0 +1,391 @@
|
|
|
1
|
+
"""Phase 9A unit 3 — pure-numpy codec-survival stage (the Phase-12-reserved home).
|
|
2
|
+
|
|
3
|
+
ARCH §2.2 / decisions 9A-A2 (G.711 μ-law/A-law PURE-NUMPY v1, ZERO new dep;
|
|
4
|
+
Opus-NB/AMR a post-v1 build-dep extra, auto-skip), 9A-A3 (module home), 9A-A7
|
|
5
|
+
(no neural codec; registry-extensible), 9A-A11 (default-ON at rung-2), 9A-A12
|
|
6
|
+
(facade ``score_codec_survival`` / ``CodecUnsupportedError``), 9A-A13 (computed
|
|
7
|
+
``phone_survival`` 4 frozen + 3 computed fields), 9A-D4.
|
|
8
|
+
|
|
9
|
+
Imports: numpy + STDLIB ONLY. G.711 μ-law/A-law are vectorized numpy companding
|
|
10
|
+
tables — NOT ``audioop`` (deprecated 3.11, REMOVED 3.13 per PEP 594; the kit's
|
|
11
|
+
dev interpreter is 3.14 → ``audioop`` cannot back it). 8 kHz resample is
|
|
12
|
+
pure-numpy decimation (no scipy). Gilbert-Elliott packet loss is seeded numpy.
|
|
13
|
+
The codec/packet operators raise on text-rung input exactly as the ``_perturb``
|
|
14
|
+
acoustic operators do (a codec round-trip on a transcript is a contract error).
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
from typing import Any
|
|
20
|
+
|
|
21
|
+
import numpy as np
|
|
22
|
+
|
|
23
|
+
# --- closed-vocabulary constants (mirrored in trinity.py, cross-pinned by a
|
|
24
|
+
# unit test — the GUNA_AXES cross-pin pattern; trinity is the gate-pinned home).
|
|
25
|
+
V1_VOICE_CODECS = ("g711_ulaw", "g711_alaw", "opus_nb", "amr_nb")
|
|
26
|
+
# g711_* = v1 pure-numpy; opus_nb/amr_nb = post-v1 build-dep, auto-skip.
|
|
27
|
+
_V1_NUMPY_CODECS = ("g711_ulaw", "g711_alaw")
|
|
28
|
+
_POST_V1_CODECS = ("opus_nb", "amr_nb")
|
|
29
|
+
V1_VOICE_PACKET_LOSS_MODELS = ("gilbert_elliott",)
|
|
30
|
+
V1_VOICE_CODEC_PROFILES = (
|
|
31
|
+
"g711_ulaw_8k_ge",
|
|
32
|
+
"g711_alaw_8k_ge",
|
|
33
|
+
"opus_nb_8k_ge",
|
|
34
|
+
"amr_nb_8k_ge",
|
|
35
|
+
"none",
|
|
36
|
+
)
|
|
37
|
+
# named bundle → (codec, packet_loss_model)
|
|
38
|
+
_PROFILE_BUNDLE = {
|
|
39
|
+
"g711_ulaw_8k_ge": ("g711_ulaw", "gilbert_elliott"),
|
|
40
|
+
"g711_alaw_8k_ge": ("g711_alaw", "gilbert_elliott"),
|
|
41
|
+
"opus_nb_8k_ge": ("opus_nb", "gilbert_elliott"),
|
|
42
|
+
"amr_nb_8k_ge": ("amr_nb", "gilbert_elliott"),
|
|
43
|
+
}
|
|
44
|
+
_TELEPHONY_RATE = 8000
|
|
45
|
+
|
|
46
|
+
# the post-v1 install path named in the auto-skip refusal (9A-A2).
|
|
47
|
+
_POST_V1_EXTRA = "agent-learning-kit[voice-codecs]"
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
class CodecUnsupportedError(RuntimeError):
|
|
51
|
+
"""Raised when a requested codec is in ``V1_VOICE_CODECS`` but its build-dep
|
|
52
|
+
extra is absent (the post-v1 ``opus_nb``/``amr_nb`` path). Callers can
|
|
53
|
+
``except CodecUnsupportedError`` to auto-skip exactly as the framework lanes
|
|
54
|
+
skip on a missing extra (the ``LANE_EXTRAS`` discipline). G.711 / packet-loss
|
|
55
|
+
never raise (numpy, always available)."""
|
|
56
|
+
|
|
57
|
+
def __init__(self, message: str, *, codec: str, install: str) -> None:
|
|
58
|
+
super().__init__(message)
|
|
59
|
+
self.codec = codec
|
|
60
|
+
self.install = install
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _require_pcm(pcm: Any, *, where: str) -> np.ndarray:
|
|
64
|
+
"""Type-guard the input as numpy PCM; a text/str input raises a ValueError
|
|
65
|
+
mirroring ``_perturb.py``'s rung-wall message (the §3.4 generalization)."""
|
|
66
|
+
|
|
67
|
+
if isinstance(pcm, (str, bytes)):
|
|
68
|
+
raise ValueError(
|
|
69
|
+
f"{where} needs a real audio channel (rung 2 loopback transport or "
|
|
70
|
+
"above); a text/transcript input is a contract error"
|
|
71
|
+
)
|
|
72
|
+
arr = np.asarray(pcm, dtype=np.float32)
|
|
73
|
+
if arr.ndim != 1:
|
|
74
|
+
arr = arr.reshape(-1)
|
|
75
|
+
return arr
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def resample_8k(pcm: np.ndarray, *, source_rate: int) -> np.ndarray:
|
|
79
|
+
"""24 kHz → 8 kHz telephony band-limit via pure-numpy anti-alias + decimation
|
|
80
|
+
(NO scipy). Anti-alias by a simple moving-average low-pass at the target
|
|
81
|
+
Nyquist (4 kHz), then decimate to ``target_rate=8000``. Deterministic."""
|
|
82
|
+
|
|
83
|
+
samples = _require_pcm(pcm, where="resample_8k")
|
|
84
|
+
if source_rate <= 0:
|
|
85
|
+
raise ValueError("source_rate must be positive")
|
|
86
|
+
if samples.size == 0 or source_rate == _TELEPHONY_RATE:
|
|
87
|
+
return samples
|
|
88
|
+
factor = source_rate / float(_TELEPHONY_RATE)
|
|
89
|
+
if factor <= 1.0:
|
|
90
|
+
# upsampling is out of scope for the telephony band-limit; pass through.
|
|
91
|
+
return samples
|
|
92
|
+
# anti-alias: moving average window ~ decimation factor (cuts > 4 kHz energy)
|
|
93
|
+
window = max(int(round(factor)), 1)
|
|
94
|
+
if window > 1:
|
|
95
|
+
kernel = np.ones(window, dtype=np.float32) / float(window)
|
|
96
|
+
samples = np.convolve(samples, kernel, mode="same").astype(np.float32)
|
|
97
|
+
n_out = int(samples.size / factor)
|
|
98
|
+
if n_out <= 0:
|
|
99
|
+
return np.zeros(0, dtype=np.float32)
|
|
100
|
+
idx = (np.arange(n_out, dtype=np.float64) * factor).astype(np.int64)
|
|
101
|
+
idx = np.clip(idx, 0, samples.size - 1)
|
|
102
|
+
return samples[idx].astype(np.float32, copy=False)
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
# --- G.711 μ-law companding (vectorized numpy, ITU-T G.711) -----------------
|
|
106
|
+
_MU = 255.0
|
|
107
|
+
_ALAW_A = 87.6
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def g711_ulaw_roundtrip(pcm: np.ndarray) -> np.ndarray:
|
|
111
|
+
"""μ-law companding round-trip via vectorized numpy (NOT audioop): encode
|
|
112
|
+
(linear → μ-law 8-bit code) then decode (μ-law code → linear). Lossy and
|
|
113
|
+
deterministic — the v1 telephony codec."""
|
|
114
|
+
|
|
115
|
+
x = _require_pcm(pcm, where="g711_ulaw_roundtrip")
|
|
116
|
+
if x.size == 0:
|
|
117
|
+
return x
|
|
118
|
+
x = np.clip(x, -1.0, 1.0)
|
|
119
|
+
# encode: μ-law compression → quantize to 8-bit code
|
|
120
|
+
sign = np.sign(x)
|
|
121
|
+
magnitude = np.log1p(_MU * np.abs(x)) / np.log1p(_MU)
|
|
122
|
+
code = np.round(magnitude * 127.0).astype(np.int32) # 8-bit magnitude quant
|
|
123
|
+
code = np.clip(code, 0, 127)
|
|
124
|
+
# decode: μ-law expansion of the quantized code
|
|
125
|
+
mag_q = code.astype(np.float32) / 127.0
|
|
126
|
+
decoded = sign * (np.expm1(mag_q * np.log1p(_MU)) / _MU)
|
|
127
|
+
return decoded.astype(np.float32, copy=False)
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def g711_alaw_roundtrip(pcm: np.ndarray) -> np.ndarray:
|
|
131
|
+
"""A-law companding round-trip, same shape as the μ-law path."""
|
|
132
|
+
|
|
133
|
+
x = _require_pcm(pcm, where="g711_alaw_roundtrip")
|
|
134
|
+
if x.size == 0:
|
|
135
|
+
return x
|
|
136
|
+
x = np.clip(x, -1.0, 1.0)
|
|
137
|
+
sign = np.sign(x)
|
|
138
|
+
ax = np.abs(x)
|
|
139
|
+
ln_a = 1.0 + np.log(_ALAW_A)
|
|
140
|
+
low = ax < (1.0 / _ALAW_A)
|
|
141
|
+
compressed = np.where(
|
|
142
|
+
low,
|
|
143
|
+
(_ALAW_A * ax) / ln_a,
|
|
144
|
+
(1.0 + np.log(np.clip(_ALAW_A * ax, 1e-12, None))) / ln_a,
|
|
145
|
+
)
|
|
146
|
+
code = np.clip(np.round(compressed * 127.0).astype(np.int32), 0, 127)
|
|
147
|
+
comp_q = code.astype(np.float32) / 127.0
|
|
148
|
+
# A-law expansion (inverse)
|
|
149
|
+
thresh = 1.0 / ln_a
|
|
150
|
+
decoded_mag = np.where(
|
|
151
|
+
comp_q < thresh,
|
|
152
|
+
(comp_q * ln_a) / _ALAW_A,
|
|
153
|
+
np.exp(comp_q * ln_a - 1.0) / _ALAW_A,
|
|
154
|
+
)
|
|
155
|
+
decoded = sign * decoded_mag
|
|
156
|
+
return decoded.astype(np.float32, copy=False)
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def gilbert_elliott_loss(
|
|
160
|
+
pcm: np.ndarray,
|
|
161
|
+
*,
|
|
162
|
+
loss_avg: float = 0.02,
|
|
163
|
+
burst_ms: float = 100.0,
|
|
164
|
+
sample_rate: int = _TELEPHONY_RATE,
|
|
165
|
+
seed: int,
|
|
166
|
+
) -> tuple[np.ndarray, dict]:
|
|
167
|
+
"""Two-state burst packet loss (the τ-Voice default 2 %/100 ms recipe,
|
|
168
|
+
R§2.2). Pure-numpy, seeded via ``np.random.default_rng(seed)`` →
|
|
169
|
+
reproducible under seed. Returns the degraded PCM + a record
|
|
170
|
+
``{model, loss_avg, burst_ms, seed, loss_realized}`` for the artifact/replay.
|
|
171
|
+
"""
|
|
172
|
+
|
|
173
|
+
x = _require_pcm(pcm, where="gilbert_elliott_loss")
|
|
174
|
+
if x.size == 0 or loss_avg <= 0:
|
|
175
|
+
return x, {
|
|
176
|
+
"model": "gilbert_elliott",
|
|
177
|
+
"loss_avg": float(loss_avg),
|
|
178
|
+
"burst_ms": float(burst_ms),
|
|
179
|
+
"seed": int(seed),
|
|
180
|
+
"loss_realized": 0.0,
|
|
181
|
+
}
|
|
182
|
+
rng = np.random.default_rng(seed)
|
|
183
|
+
frame_samples = max(int(sample_rate * 20.0 / 1000.0), 1) # 20 ms frames
|
|
184
|
+
n_frames = max(int(np.ceil(x.size / frame_samples)), 1)
|
|
185
|
+
# two-state Markov chain: G(ood) and B(ad). Mean burst length = burst_ms/20ms
|
|
186
|
+
# frames ⇒ p(B→G) = 1/burst_frames; steady-state loss = loss_avg ⇒ derive
|
|
187
|
+
# p(G→B) from the balance equation π_B = loss_avg.
|
|
188
|
+
burst_frames = max(burst_ms / 20.0, 1.0)
|
|
189
|
+
p_bg = 1.0 / burst_frames # leave Bad
|
|
190
|
+
# π_B = p_gb / (p_gb + p_bg) = loss_avg ⇒ p_gb = loss_avg * p_bg / (1-loss_avg)
|
|
191
|
+
p_gb = (loss_avg * p_bg) / max(1.0 - loss_avg, 1e-9)
|
|
192
|
+
degraded = x.copy()
|
|
193
|
+
bad = False
|
|
194
|
+
lost_frames = 0
|
|
195
|
+
for f in range(n_frames):
|
|
196
|
+
if bad:
|
|
197
|
+
if rng.random() < p_bg:
|
|
198
|
+
bad = False
|
|
199
|
+
else:
|
|
200
|
+
if rng.random() < p_gb:
|
|
201
|
+
bad = True
|
|
202
|
+
if bad:
|
|
203
|
+
start = f * frame_samples
|
|
204
|
+
degraded[start : start + frame_samples] = 0.0
|
|
205
|
+
lost_frames += 1
|
|
206
|
+
record = {
|
|
207
|
+
"model": "gilbert_elliott",
|
|
208
|
+
"loss_avg": float(loss_avg),
|
|
209
|
+
"burst_ms": float(burst_ms),
|
|
210
|
+
"seed": int(seed),
|
|
211
|
+
"loss_realized": round(lost_frames / float(n_frames), 6),
|
|
212
|
+
}
|
|
213
|
+
return degraded.astype(np.float32, copy=False), record
|
|
214
|
+
|
|
215
|
+
|
|
216
|
+
def _codec_roundtrip(pcm: np.ndarray, *, codec: str) -> np.ndarray:
|
|
217
|
+
if codec == "g711_ulaw":
|
|
218
|
+
return g711_ulaw_roundtrip(pcm)
|
|
219
|
+
if codec == "g711_alaw":
|
|
220
|
+
return g711_alaw_roundtrip(pcm)
|
|
221
|
+
if codec in _POST_V1_CODECS:
|
|
222
|
+
raise CodecUnsupportedError(
|
|
223
|
+
f"codec {codec!r} is a post-v1 build-dep extra and is not installed; "
|
|
224
|
+
f"install {_POST_V1_EXTRA} to enable it (v1 ships g711_ulaw/g711_alaw "
|
|
225
|
+
"as the required pure-numpy codecs)",
|
|
226
|
+
codec=codec,
|
|
227
|
+
install=_POST_V1_EXTRA,
|
|
228
|
+
)
|
|
229
|
+
raise ValueError(f"unknown codec {codec!r}; expected one of {V1_VOICE_CODECS}")
|
|
230
|
+
|
|
231
|
+
|
|
232
|
+
def _apply_channel(
|
|
233
|
+
pcm: np.ndarray, *, codec: str, packet_loss: str, seed: int, sample_rate: int
|
|
234
|
+
) -> tuple[np.ndarray, dict]:
|
|
235
|
+
"""Resample → codec round-trip → packet loss, returning degraded PCM + the
|
|
236
|
+
codec_round_trip record (UI-UX §3.3 shape)."""
|
|
237
|
+
|
|
238
|
+
if packet_loss not in V1_VOICE_PACKET_LOSS_MODELS:
|
|
239
|
+
raise ValueError(
|
|
240
|
+
f"packet_loss_model {packet_loss!r} must be one of "
|
|
241
|
+
f"{V1_VOICE_PACKET_LOSS_MODELS}"
|
|
242
|
+
)
|
|
243
|
+
resampled = resample_8k(pcm, source_rate=sample_rate)
|
|
244
|
+
coded = _codec_roundtrip(resampled, codec=codec)
|
|
245
|
+
degraded, loss_record = gilbert_elliott_loss(
|
|
246
|
+
coded, sample_rate=_TELEPHONY_RATE, seed=seed
|
|
247
|
+
)
|
|
248
|
+
record = {
|
|
249
|
+
"codec": codec,
|
|
250
|
+
"resampled_to_hz": _TELEPHONY_RATE,
|
|
251
|
+
"source_rate_hz": int(sample_rate),
|
|
252
|
+
"packet_loss": loss_record,
|
|
253
|
+
"seed": int(seed),
|
|
254
|
+
}
|
|
255
|
+
return degraded, record
|
|
256
|
+
|
|
257
|
+
|
|
258
|
+
def apply_codec_profile(
|
|
259
|
+
user_pcm: np.ndarray,
|
|
260
|
+
agent_pcm: np.ndarray,
|
|
261
|
+
*,
|
|
262
|
+
profile: str,
|
|
263
|
+
seed: int,
|
|
264
|
+
sample_rate: int,
|
|
265
|
+
) -> tuple[np.ndarray, np.ndarray, dict]:
|
|
266
|
+
"""Apply a named codec profile (codec + resample + packet-loss bundle) to
|
|
267
|
+
both streams. ``profile='none'`` is a no-op (clean-PCM loopback). Raises
|
|
268
|
+
``CodecUnsupportedError`` for ``opus_nb_8k_ge``/``amr_nb_8k_ge`` when the
|
|
269
|
+
extra is absent (post-v1). Returns degraded (user_pcm, agent_pcm) + the
|
|
270
|
+
codec_round_trip record (UI-UX §3.3 shape)."""
|
|
271
|
+
|
|
272
|
+
if profile not in V1_VOICE_CODEC_PROFILES:
|
|
273
|
+
raise ValueError(
|
|
274
|
+
f"codec_profile {profile!r} must be one of {V1_VOICE_CODEC_PROFILES}"
|
|
275
|
+
)
|
|
276
|
+
if profile == "none":
|
|
277
|
+
return (
|
|
278
|
+
_require_pcm(user_pcm, where="apply_codec_profile"),
|
|
279
|
+
_require_pcm(agent_pcm, where="apply_codec_profile"),
|
|
280
|
+
{"profile": "none", "applied": False},
|
|
281
|
+
)
|
|
282
|
+
codec, packet_loss = _PROFILE_BUNDLE[profile]
|
|
283
|
+
user_deg, user_rec = _apply_channel(
|
|
284
|
+
user_pcm, codec=codec, packet_loss=packet_loss, seed=seed, sample_rate=sample_rate
|
|
285
|
+
)
|
|
286
|
+
agent_deg, agent_rec = _apply_channel(
|
|
287
|
+
agent_pcm,
|
|
288
|
+
codec=codec,
|
|
289
|
+
packet_loss=packet_loss,
|
|
290
|
+
seed=seed + 1,
|
|
291
|
+
sample_rate=sample_rate,
|
|
292
|
+
)
|
|
293
|
+
record = {
|
|
294
|
+
"profile": profile,
|
|
295
|
+
"applied": True,
|
|
296
|
+
"codec": codec,
|
|
297
|
+
"packet_loss_model": packet_loss,
|
|
298
|
+
"resampled_to_hz": _TELEPHONY_RATE,
|
|
299
|
+
"source_rate_hz": int(sample_rate),
|
|
300
|
+
"user": user_rec,
|
|
301
|
+
"agent": agent_rec,
|
|
302
|
+
"seed": int(seed),
|
|
303
|
+
}
|
|
304
|
+
return user_deg, agent_deg, record
|
|
305
|
+
|
|
306
|
+
|
|
307
|
+
def _band_energy_lt_4khz(pcm: np.ndarray, *, sample_rate: int) -> float:
|
|
308
|
+
"""Fraction of signal energy below 4 kHz the telephony codec preserves
|
|
309
|
+
(CodecAttack <4 kHz framing, R§2.2)."""
|
|
310
|
+
|
|
311
|
+
x = _require_pcm(pcm, where="band_energy")
|
|
312
|
+
if x.size < 2:
|
|
313
|
+
return 0.0
|
|
314
|
+
spectrum = np.abs(np.fft.rfft(x)) ** 2
|
|
315
|
+
freqs = np.fft.rfftfreq(x.size, d=1.0 / float(sample_rate))
|
|
316
|
+
total = float(spectrum.sum())
|
|
317
|
+
if total <= 0:
|
|
318
|
+
return 0.0
|
|
319
|
+
low = float(spectrum[freqs < 4000.0].sum())
|
|
320
|
+
return round(low / total, 6)
|
|
321
|
+
|
|
322
|
+
|
|
323
|
+
def _success_score(pcm: np.ndarray) -> float:
|
|
324
|
+
"""A reproducible proxy success score for the channel pre/post twins: RMS
|
|
325
|
+
energy normalized to [0,1]. The clean twin scores higher than the degraded
|
|
326
|
+
twin (the channel attenuates / drops frames), so the pre→post delta is the
|
|
327
|
+
survival evidence. Deterministic — no model call."""
|
|
328
|
+
|
|
329
|
+
x = _require_pcm(pcm, where="success_score")
|
|
330
|
+
if x.size == 0:
|
|
331
|
+
return 0.0
|
|
332
|
+
rms = float(np.sqrt((x.astype(np.float64) ** 2).mean()))
|
|
333
|
+
return round(min(rms * 2.0, 1.0), 6)
|
|
334
|
+
|
|
335
|
+
|
|
336
|
+
def score_codec_survival(
|
|
337
|
+
user_pcm: np.ndarray,
|
|
338
|
+
agent_pcm: np.ndarray,
|
|
339
|
+
*,
|
|
340
|
+
codec: str,
|
|
341
|
+
packet_loss: str,
|
|
342
|
+
seed: int,
|
|
343
|
+
sample_rate: int = 24000,
|
|
344
|
+
pre_channel_success: float | None = None,
|
|
345
|
+
) -> dict:
|
|
346
|
+
"""Re-validate an acoustic claim through the telephony channel; return the
|
|
347
|
+
COMPUTED ``phone_survival`` object (9A-A13). The Phase-12 frozen 4 fields
|
|
348
|
+
(``status``/``tier``/``scope_label?``/``reason``) keep their vocabulary
|
|
349
|
+
unchanged; 3 OPTIONAL computed-evidence fields
|
|
350
|
+
(``pre_channel_success``/``post_channel_success``/``band_energy_lt_4khz``)
|
|
351
|
+
are present ONLY when ``tier ∈ {channel_simulated, channel_live}``."""
|
|
352
|
+
|
|
353
|
+
_require_pcm(user_pcm, where="score_codec_survival") # type-guard the user side
|
|
354
|
+
agent = _require_pcm(agent_pcm, where="score_codec_survival")
|
|
355
|
+
# the clean twin success (BEFORE the channel) — measured on the agent side
|
|
356
|
+
# (the side carrying the claim under test) unless supplied by the caller.
|
|
357
|
+
pre = (
|
|
358
|
+
float(pre_channel_success)
|
|
359
|
+
if pre_channel_success is not None
|
|
360
|
+
else _success_score(agent)
|
|
361
|
+
)
|
|
362
|
+
agent_deg, channel_record = _apply_channel(
|
|
363
|
+
agent, codec=codec, packet_loss=packet_loss, seed=seed, sample_rate=sample_rate
|
|
364
|
+
)
|
|
365
|
+
post = _success_score(agent_deg)
|
|
366
|
+
band = _band_energy_lt_4khz(agent_deg, sample_rate=_TELEPHONY_RATE)
|
|
367
|
+
|
|
368
|
+
# status derives from the pre→post delta
|
|
369
|
+
if pre <= 0:
|
|
370
|
+
status = "untested"
|
|
371
|
+
else:
|
|
372
|
+
retained = post / pre if pre > 0 else 0.0
|
|
373
|
+
if retained >= 0.85:
|
|
374
|
+
status = "survives"
|
|
375
|
+
elif retained >= 0.4:
|
|
376
|
+
status = "partial"
|
|
377
|
+
else:
|
|
378
|
+
status = "dies"
|
|
379
|
+
|
|
380
|
+
return {
|
|
381
|
+
"status": status,
|
|
382
|
+
"tier": "channel_simulated",
|
|
383
|
+
"reason": (
|
|
384
|
+
f"codec={codec} packet_loss={packet_loss} "
|
|
385
|
+
f"loss_realized={channel_record['packet_loss']['loss_realized']} "
|
|
386
|
+
f"pre={pre} post={post} (8 kHz telephony channel, simulated)"
|
|
387
|
+
),
|
|
388
|
+
"pre_channel_success": round(pre, 6),
|
|
389
|
+
"post_channel_success": round(post, 6),
|
|
390
|
+
"band_energy_lt_4khz": band,
|
|
391
|
+
}
|
fi/alk/live/_contract.py
ADDED
|
@@ -0,0 +1,134 @@
|
|
|
1
|
+
"""Live-lane contract: vocabularies, lane specs, budgets, flag discipline.
|
|
2
|
+
|
|
3
|
+
Imports: stdlib only. Every substrate module must remain importable (and
|
|
4
|
+
its unit tests green) in an environment with no framework extra installed —
|
|
5
|
+
the live_lane_boundary gate scans them like any other release module.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import dataclasses
|
|
11
|
+
import os
|
|
12
|
+
from typing import Any, Mapping
|
|
13
|
+
|
|
14
|
+
# --- artifact kind (same kind simulate/cli emit — never a parallel kind) ----
|
|
15
|
+
AGENT_LEARNING_RUN_KIND = "agent-learning.run.v1"
|
|
16
|
+
|
|
17
|
+
# --- evidence classes (R§3.2; PRD §4.1) -----------------------------------
|
|
18
|
+
EVIDENCE_CLASSES = ("local_gate", "live_lane", "live_stressed", "captured_fixture")
|
|
19
|
+
RELEASE_ADMISSIBLE_EVIDENCE_CLASSES = ("local_gate", "captured_fixture")
|
|
20
|
+
|
|
21
|
+
# --- failure layers (R§1 #1 HarnessFix; PRD §4.1) --------------------------
|
|
22
|
+
FAILURE_LAYERS = ("lane_infra", "framework_runtime", "provider", "agent_behavior")
|
|
23
|
+
|
|
24
|
+
# --- per-scenario verdicts (R§3.4) ------------------------------------------
|
|
25
|
+
VERDICTS = ("pass", "fail", "unstable", "void")
|
|
26
|
+
|
|
27
|
+
# --- env-flag conventions (PRD §4.1: AGENT_LEARNING_LIVE_<LANE>=1) ---------
|
|
28
|
+
LANE_ENV_FLAGS = {
|
|
29
|
+
"livekit": "AGENT_LEARNING_LIVE_LIVEKIT",
|
|
30
|
+
"pipecat": "AGENT_LEARNING_LIVE_PIPECAT",
|
|
31
|
+
"langchain": "AGENT_LEARNING_LIVE_LANGCHAIN",
|
|
32
|
+
"mcp": "AGENT_LEARNING_LIVE_MCP",
|
|
33
|
+
"a2a": "AGENT_LEARNING_LIVE_A2A",
|
|
34
|
+
"credentialed": "AGENT_LEARNING_LIVE_CREDENTIALED",
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
# --- lane → extra map (skip lines and import errors name these) ------------
|
|
38
|
+
LANE_EXTRAS = {
|
|
39
|
+
"livekit": "livekit",
|
|
40
|
+
"pipecat": "pipecat",
|
|
41
|
+
"langchain": "langchain",
|
|
42
|
+
"mcp": "mcp",
|
|
43
|
+
"a2a": "a2a",
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
# --- budget caps (P3-D2): 600 s default; voice lanes 900 s -----------------
|
|
47
|
+
LANE_BUDGET_S_DEFAULT = 600.0
|
|
48
|
+
LANE_BUDGET_S = {"livekit": 900.0, "pipecat": 900.0}
|
|
49
|
+
|
|
50
|
+
# --- repeat policy (P3-D2) ---------------------------------------------------
|
|
51
|
+
DEFAULT_REPEATS = 8
|
|
52
|
+
UNSTABLE_ICC_FLOOR = 0.5
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def lane_budget_s(lane: str) -> float:
|
|
56
|
+
"""Hard wall-clock cap for one lane run (P3-D2)."""
|
|
57
|
+
|
|
58
|
+
return LANE_BUDGET_S.get(lane, LANE_BUDGET_S_DEFAULT)
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
class LaneDisabledError(RuntimeError):
|
|
62
|
+
"""Raised when a lane entry point runs without its env flag."""
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def require_lane_enabled(lane: str) -> None:
|
|
66
|
+
"""Gate every lane entry on its env flag (PRD §4.1).
|
|
67
|
+
|
|
68
|
+
The live_lane_boundary gate statically asserts every public lane module
|
|
69
|
+
calls this (unit 4.2 check 3) — the dynamic raise and the static scan
|
|
70
|
+
are two halves of the same discipline.
|
|
71
|
+
"""
|
|
72
|
+
|
|
73
|
+
flag = LANE_ENV_FLAGS[lane]
|
|
74
|
+
if os.environ.get(flag) != "1":
|
|
75
|
+
raise LaneDisabledError(
|
|
76
|
+
f"live lane '{lane}' is opt-in: set {flag}=1 to run it "
|
|
77
|
+
"(never set in release flows)"
|
|
78
|
+
)
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
@dataclasses.dataclass(frozen=True)
|
|
82
|
+
class LaneSpec:
|
|
83
|
+
"""What a lane run was asked to do — shared by runner and lanes."""
|
|
84
|
+
|
|
85
|
+
lane: str
|
|
86
|
+
scenario: Mapping[str, Any]
|
|
87
|
+
rung: int = 1
|
|
88
|
+
required_env: tuple[str, ...] = ()
|
|
89
|
+
version_requirement: str | None = None
|
|
90
|
+
repeats: int = DEFAULT_REPEATS
|
|
91
|
+
budget_s: float | None = None
|
|
92
|
+
evidence_class: str = "live_lane"
|
|
93
|
+
|
|
94
|
+
def resolved_budget_s(self) -> float:
|
|
95
|
+
if self.budget_s is not None:
|
|
96
|
+
return float(self.budget_s)
|
|
97
|
+
return lane_budget_s(self.lane)
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
@dataclasses.dataclass
|
|
101
|
+
class LaneRun:
|
|
102
|
+
"""One repeat of one scenario inside a lane run (a per_repeat row)."""
|
|
103
|
+
|
|
104
|
+
index: int
|
|
105
|
+
passed: bool | None
|
|
106
|
+
score: float | None
|
|
107
|
+
failure_layer: str | None
|
|
108
|
+
quarantined: bool
|
|
109
|
+
evidence_class: str
|
|
110
|
+
detail: str = ""
|
|
111
|
+
void_reason: str | None = None
|
|
112
|
+
transcript_path: str | None = None
|
|
113
|
+
transcript_complete: bool | None = None
|
|
114
|
+
transcript_sha256: str | None = None
|
|
115
|
+
step_signature: tuple[str, ...] = ()
|
|
116
|
+
|
|
117
|
+
def to_row(self) -> dict[str, Any]:
|
|
118
|
+
row: dict[str, Any] = {
|
|
119
|
+
"repeat": self.index,
|
|
120
|
+
"passed": self.passed,
|
|
121
|
+
"score": self.score,
|
|
122
|
+
"failure_layer": self.failure_layer,
|
|
123
|
+
"quarantined": self.quarantined,
|
|
124
|
+
"evidence_class": self.evidence_class,
|
|
125
|
+
"transcript_path": self.transcript_path,
|
|
126
|
+
"transcript_complete": self.transcript_complete,
|
|
127
|
+
"transcript_sha256": self.transcript_sha256,
|
|
128
|
+
"step_signature": list(self.step_signature),
|
|
129
|
+
}
|
|
130
|
+
if self.detail:
|
|
131
|
+
row["detail"] = self.detail
|
|
132
|
+
if self.void_reason:
|
|
133
|
+
row["void_reason"] = self.void_reason
|
|
134
|
+
return row
|