agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
fi/alk/live/_stats.py
ADDED
|
@@ -0,0 +1,561 @@
|
|
|
1
|
+
"""Variance math, repeat executor, verdicts for live lanes — pure numpy.
|
|
2
|
+
|
|
3
|
+
ICC(1) via one-way variance decomposition (ARCH Decision 4 — no scipy);
|
|
4
|
+
degenerate zero-variance matrices define ICC := 1.0 (a deterministic green
|
|
5
|
+
run must never classify ``unstable``). Determinism metrics are reported
|
|
6
|
+
separately from quality scores — DFAH's r=-0.11 forbids conflation (R§1 #15).
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import dataclasses
|
|
12
|
+
import math
|
|
13
|
+
import tempfile
|
|
14
|
+
import time
|
|
15
|
+
import uuid
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
from typing import Any, Callable, Mapping, Sequence
|
|
18
|
+
|
|
19
|
+
import numpy as np
|
|
20
|
+
|
|
21
|
+
from .._schema import public_payload
|
|
22
|
+
from ._contract import (
|
|
23
|
+
AGENT_LEARNING_RUN_KIND,
|
|
24
|
+
DEFAULT_REPEATS,
|
|
25
|
+
EVIDENCE_CLASSES,
|
|
26
|
+
FAILURE_LAYERS,
|
|
27
|
+
UNSTABLE_ICC_FLOOR,
|
|
28
|
+
LaneRun,
|
|
29
|
+
lane_budget_s,
|
|
30
|
+
)
|
|
31
|
+
from ._transcript import TranscriptRecorder
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def icc_and_within_variance(scores: np.ndarray) -> tuple[float, float]:
|
|
35
|
+
"""One-way random-effects ICC over a (n_scenarios, k_repeats) score
|
|
36
|
+
matrix + pooled within-scenario variance (R§1 #16, ICC convergence at
|
|
37
|
+
n=8–16; P3-D2 default k=8).
|
|
38
|
+
|
|
39
|
+
ICC = (MS_between - MS_within) / (MS_between + (k-1)·MS_within)
|
|
40
|
+
"""
|
|
41
|
+
|
|
42
|
+
scores = np.asarray(scores, dtype=float)
|
|
43
|
+
if scores.ndim != 2:
|
|
44
|
+
raise ValueError("scores must be a 2-D (n_scenarios, k_repeats) matrix")
|
|
45
|
+
n, k = scores.shape
|
|
46
|
+
grand = scores.mean()
|
|
47
|
+
row_means = scores.mean(axis=1)
|
|
48
|
+
ms_between = k * ((row_means - grand) ** 2).sum() / max(n - 1, 1)
|
|
49
|
+
ms_within = ((scores - row_means[:, None]) ** 2).sum() / max(n * (k - 1), 1)
|
|
50
|
+
denominator = ms_between + (k - 1) * ms_within
|
|
51
|
+
if denominator == 0:
|
|
52
|
+
# Degenerate zero-variance matrix (e.g. an all-pass run with
|
|
53
|
+
# byte-identical scores): perfect consistency by definition —
|
|
54
|
+
# ICC := 1.0. Without this rule a perfectly deterministic green
|
|
55
|
+
# run would classify `unstable` (review finding F2).
|
|
56
|
+
return 1.0, float(ms_within)
|
|
57
|
+
return float((ms_between - ms_within) / denominator), float(ms_within)
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def divergence_step(step_signatures: Sequence[Sequence[str]]) -> int | None:
|
|
61
|
+
"""First step index at which repeated trajectories fork (R§1 #17 — 69%
|
|
62
|
+
fork at step 2, so this is cheap, high-signal evidence). Signatures are
|
|
63
|
+
normalized step strings (tool name + outcome class, no payloads).
|
|
64
|
+
Returns None when all repeats share one trajectory."""
|
|
65
|
+
|
|
66
|
+
longest = max((len(s) for s in step_signatures), default=0)
|
|
67
|
+
for index in range(longest):
|
|
68
|
+
prefixes = {tuple(s[: index + 1]) for s in step_signatures}
|
|
69
|
+
if len(prefixes) > 1:
|
|
70
|
+
return index
|
|
71
|
+
return None
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def determinism_metrics(
|
|
75
|
+
step_signatures: Sequence[Sequence[str]],
|
|
76
|
+
) -> dict[str, Any]:
|
|
77
|
+
"""Trajectory-distribution metrics, kept strictly separate from quality
|
|
78
|
+
(R§1 #15). Entropy is Shannon entropy in bits over distinct trajectories."""
|
|
79
|
+
|
|
80
|
+
trajectories = [tuple(signature) for signature in step_signatures]
|
|
81
|
+
if not trajectories:
|
|
82
|
+
return {"distinct_trajectory_count": 0, "trajectory_entropy": 0.0}
|
|
83
|
+
counts: dict[tuple[str, ...], int] = {}
|
|
84
|
+
for trajectory in trajectories:
|
|
85
|
+
counts[trajectory] = counts.get(trajectory, 0) + 1
|
|
86
|
+
total = len(trajectories)
|
|
87
|
+
entropy = -sum(
|
|
88
|
+
(count / total) * math.log2(count / total) for count in counts.values()
|
|
89
|
+
)
|
|
90
|
+
return {
|
|
91
|
+
"distinct_trajectory_count": len(counts),
|
|
92
|
+
"trajectory_entropy": round(float(entropy), 6),
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def step_signature_from_events(
|
|
97
|
+
events: Sequence[Mapping[str, Any]],
|
|
98
|
+
) -> list[str]:
|
|
99
|
+
"""Normalize a transcript into step strings for divergence detection:
|
|
100
|
+
tool name + outcome class, message marks — never payloads."""
|
|
101
|
+
|
|
102
|
+
signature: list[str] = []
|
|
103
|
+
for event in events:
|
|
104
|
+
channel = str(event.get("channel") or "")
|
|
105
|
+
event_type = str(event.get("type") or "")
|
|
106
|
+
payload = event.get("payload")
|
|
107
|
+
payload = payload if isinstance(payload, Mapping) else {}
|
|
108
|
+
if channel == "tool":
|
|
109
|
+
name = str(payload.get("name") or payload.get("tool") or "tool")
|
|
110
|
+
if payload.get("error") or payload.get("ok") is False:
|
|
111
|
+
outcome = "error"
|
|
112
|
+
else:
|
|
113
|
+
outcome = "ok"
|
|
114
|
+
signature.append(f"tool:{name}:{outcome}")
|
|
115
|
+
elif channel in ("user", "agent"):
|
|
116
|
+
signature.append(f"{channel}:{event_type}")
|
|
117
|
+
return signature
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
@dataclasses.dataclass
|
|
121
|
+
class LaneRunResult:
|
|
122
|
+
lane: str
|
|
123
|
+
evidence_class: str # stamped at construction; member of EVIDENCE_CLASSES
|
|
124
|
+
repeats: int
|
|
125
|
+
verdict: str # "pass" | "fail" | "unstable" | "void"
|
|
126
|
+
per_repeat: list[dict] # {score, passed, failure_layer, transcript_path}
|
|
127
|
+
icc: float | None
|
|
128
|
+
within_variance: float | None
|
|
129
|
+
divergence_step: int | None
|
|
130
|
+
determinism: dict # {distinct_trajectory_count, trajectory_entropy}
|
|
131
|
+
quarantined_repeats: int # lane_infra rows excluded from stats (R§1 #7 validate-then-score)
|
|
132
|
+
required_env: list[str] # NAMES only, never values
|
|
133
|
+
end_state_diff: dict | None # before/after snapshot (R§1 #14 Saber)
|
|
134
|
+
# --- run identity + budget mechanics (ARCH §4) — open details, simplest
|
|
135
|
+
# additive fields consistent with the architecture: ---------------------
|
|
136
|
+
run_id: str = ""
|
|
137
|
+
rung: str | int = 1
|
|
138
|
+
framework: str | None = None
|
|
139
|
+
framework_version: str | None = None
|
|
140
|
+
version_requirement: str | None = None
|
|
141
|
+
version_ok: bool | None = None
|
|
142
|
+
repeats_requested: int = 0
|
|
143
|
+
repeats_completed: int = 0
|
|
144
|
+
budget_cap_s: float = 0.0
|
|
145
|
+
budget_spent_s: float = 0.0
|
|
146
|
+
verdict_reason: str | None = None
|
|
147
|
+
findings: list[dict] = dataclasses.field(default_factory=list)
|
|
148
|
+
artifacts_dir: str | None = None
|
|
149
|
+
|
|
150
|
+
def to_block(self) -> dict[str, Any]:
|
|
151
|
+
"""The ``live_lane`` evidence block of the run.v1 payload."""
|
|
152
|
+
|
|
153
|
+
return dataclasses.asdict(self)
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def run_repeated(
|
|
157
|
+
run_once: Callable[[int, TranscriptRecorder], dict],
|
|
158
|
+
*,
|
|
159
|
+
lane: str,
|
|
160
|
+
evidence_class: str,
|
|
161
|
+
repeats: int = DEFAULT_REPEATS, # P3-D2 default; --repeats override upstream
|
|
162
|
+
budget_s: float | None = None, # None → LANE_BUDGET_S.get(lane, default):
|
|
163
|
+
# 600 s default, 900 s voice lanes (P3-D2)
|
|
164
|
+
unstable_icc_floor: float = UNSTABLE_ICC_FLOOR,
|
|
165
|
+
required_env: Sequence[str] = (),
|
|
166
|
+
artifacts_dir: str | Path | None = None,
|
|
167
|
+
run_id: str | None = None,
|
|
168
|
+
rung: str | int = 1,
|
|
169
|
+
framework: str | None = None,
|
|
170
|
+
version_requirement: str | None = None,
|
|
171
|
+
) -> LaneRunResult:
|
|
172
|
+
"""Repeat executor. lane_infra rows are quarantined (excluded from the
|
|
173
|
+
score matrix AND counted); verifier evidence is mandatory per repeat —
|
|
174
|
+
a repeat with no programmatic/judge/end-state verdict is itself
|
|
175
|
+
lane_infra (R§1 #5: sampling without verification is the documented gap).
|
|
176
|
+
|
|
177
|
+
Per-scenario verdict (R§3.4; lane-run exit policy lives in unit 6):
|
|
178
|
+
pass — every non-quarantined repeat passed AND icc >= floor
|
|
179
|
+
(zero-variance all-pass runs hit this via ICC := 1.0 above)
|
|
180
|
+
fail — every non-quarantined repeat failed
|
|
181
|
+
unstable — mixed outcomes, or icc < floor; quarantined like a flaky
|
|
182
|
+
test with fork evidence attached, never a red/green coin flip
|
|
183
|
+
void — lane_infra consumed the sample (no scoreable repeats);
|
|
184
|
+
the ONLY source of `void` (PRD §4.1).
|
|
185
|
+
"""
|
|
186
|
+
|
|
187
|
+
if evidence_class not in EVIDENCE_CLASSES:
|
|
188
|
+
raise ValueError(f"unknown evidence_class: {evidence_class!r}")
|
|
189
|
+
if repeats < 1:
|
|
190
|
+
raise ValueError("repeats must be >= 1")
|
|
191
|
+
budget = float(budget_s) if budget_s is not None else lane_budget_s(lane)
|
|
192
|
+
base_dir = (
|
|
193
|
+
Path(artifacts_dir)
|
|
194
|
+
if artifacts_dir is not None
|
|
195
|
+
else Path(tempfile.mkdtemp(prefix=f"agent-learning-live-{lane}-"))
|
|
196
|
+
)
|
|
197
|
+
base_dir.mkdir(parents=True, exist_ok=True)
|
|
198
|
+
resolved_run_id = run_id or uuid.uuid4().hex
|
|
199
|
+
started = time.monotonic()
|
|
200
|
+
|
|
201
|
+
rows: list[LaneRun] = []
|
|
202
|
+
findings: list[dict] = []
|
|
203
|
+
framework_name = framework
|
|
204
|
+
framework_version: str | None = None
|
|
205
|
+
version_ok_observed: bool | None = None
|
|
206
|
+
end_state_diff: dict | None = None
|
|
207
|
+
budget_exhausted = False
|
|
208
|
+
|
|
209
|
+
for index in range(repeats):
|
|
210
|
+
if time.monotonic() - started >= budget:
|
|
211
|
+
budget_exhausted = True
|
|
212
|
+
break
|
|
213
|
+
transcript = TranscriptRecorder(
|
|
214
|
+
base_dir / f"repeat-{index:02d}.jsonl",
|
|
215
|
+
required_env=required_env,
|
|
216
|
+
)
|
|
217
|
+
try:
|
|
218
|
+
outcome: Mapping[str, Any] = run_once(index, transcript) or {}
|
|
219
|
+
except Exception as exc: # our machinery failing is lane_infra, never a score
|
|
220
|
+
outcome = {
|
|
221
|
+
"passed": None,
|
|
222
|
+
"score": None,
|
|
223
|
+
"failure_layer": "lane_infra",
|
|
224
|
+
"void_reason": f"lane runner exception: {exc}",
|
|
225
|
+
"detail": f"lane runner exception: {exc}",
|
|
226
|
+
}
|
|
227
|
+
finally:
|
|
228
|
+
summary = transcript.close()
|
|
229
|
+
|
|
230
|
+
failure_layer = outcome.get("failure_layer")
|
|
231
|
+
if failure_layer is not None and failure_layer not in FAILURE_LAYERS:
|
|
232
|
+
failure_layer = "lane_infra"
|
|
233
|
+
passed = outcome.get("passed")
|
|
234
|
+
if passed is None and failure_layer is None:
|
|
235
|
+
# No verdict at all → the repeat itself is lane_infra (R§1 #5).
|
|
236
|
+
failure_layer = "lane_infra"
|
|
237
|
+
outcome = dict(outcome)
|
|
238
|
+
outcome.setdefault(
|
|
239
|
+
"void_reason", "no verifier evidence for this repeat"
|
|
240
|
+
)
|
|
241
|
+
outcome.setdefault(
|
|
242
|
+
"detail", "no verifier evidence for this repeat"
|
|
243
|
+
)
|
|
244
|
+
quarantined = failure_layer == "lane_infra"
|
|
245
|
+
score = outcome.get("score")
|
|
246
|
+
if score is None and not quarantined:
|
|
247
|
+
score = 1.0 if passed else 0.0
|
|
248
|
+
|
|
249
|
+
version_info = outcome.get("version")
|
|
250
|
+
if isinstance(version_info, Mapping):
|
|
251
|
+
framework_name = framework_name or version_info.get("framework")
|
|
252
|
+
framework_version = framework_version or version_info.get(
|
|
253
|
+
"framework_version"
|
|
254
|
+
)
|
|
255
|
+
if version_info.get("version_ok") is False:
|
|
256
|
+
version_ok_observed = False
|
|
257
|
+
findings.append(
|
|
258
|
+
{
|
|
259
|
+
"type": "live_lane_framework_version_mismatch",
|
|
260
|
+
"level": "error",
|
|
261
|
+
"repeat": index,
|
|
262
|
+
"detail": version_info.get("void_reason"),
|
|
263
|
+
}
|
|
264
|
+
)
|
|
265
|
+
elif version_ok_observed is None:
|
|
266
|
+
version_ok_observed = bool(version_info.get("version_ok"))
|
|
267
|
+
if isinstance(outcome.get("end_state_diff"), Mapping):
|
|
268
|
+
end_state_diff = dict(outcome["end_state_diff"])
|
|
269
|
+
|
|
270
|
+
if not summary.get("complete", True):
|
|
271
|
+
findings.append(
|
|
272
|
+
{
|
|
273
|
+
"type": "live_lane_transcript_truncated",
|
|
274
|
+
"level": "warning",
|
|
275
|
+
"repeat": index,
|
|
276
|
+
"detail": summary.get("truncated"),
|
|
277
|
+
}
|
|
278
|
+
)
|
|
279
|
+
|
|
280
|
+
rows.append(
|
|
281
|
+
LaneRun(
|
|
282
|
+
index=index,
|
|
283
|
+
passed=None if quarantined else bool(passed),
|
|
284
|
+
score=None if quarantined else float(score),
|
|
285
|
+
failure_layer=failure_layer,
|
|
286
|
+
quarantined=quarantined,
|
|
287
|
+
evidence_class=evidence_class,
|
|
288
|
+
detail=str(outcome.get("detail") or ""),
|
|
289
|
+
void_reason=outcome.get("void_reason"),
|
|
290
|
+
transcript_path=str(summary.get("path")),
|
|
291
|
+
transcript_complete=bool(summary.get("complete", True)),
|
|
292
|
+
transcript_sha256=summary.get("sha256"),
|
|
293
|
+
step_signature=tuple(outcome.get("step_signature") or ()),
|
|
294
|
+
)
|
|
295
|
+
)
|
|
296
|
+
|
|
297
|
+
budget_spent = time.monotonic() - started
|
|
298
|
+
scoreable = [row for row in rows if not row.quarantined]
|
|
299
|
+
quarantined_count = sum(1 for row in rows if row.quarantined)
|
|
300
|
+
|
|
301
|
+
if scoreable:
|
|
302
|
+
matrix = np.asarray([[row.score for row in scoreable]], dtype=float)
|
|
303
|
+
icc, within = icc_and_within_variance(matrix)
|
|
304
|
+
else:
|
|
305
|
+
icc, within = None, None
|
|
306
|
+
|
|
307
|
+
signatures = [
|
|
308
|
+
list(row.step_signature) for row in scoreable if row.step_signature
|
|
309
|
+
]
|
|
310
|
+
fork_step = divergence_step(signatures) if signatures else None
|
|
311
|
+
determinism = determinism_metrics(signatures)
|
|
312
|
+
|
|
313
|
+
verdict_reason: str | None = None
|
|
314
|
+
if not scoreable:
|
|
315
|
+
verdict = "void"
|
|
316
|
+
verdict_reason = "lane_infra_consumed_sample"
|
|
317
|
+
findings.append(
|
|
318
|
+
{
|
|
319
|
+
"type": "live_lane_infra_void",
|
|
320
|
+
"level": "error",
|
|
321
|
+
"detail": "lane_infra consumed the sample (no scoreable repeats)",
|
|
322
|
+
}
|
|
323
|
+
)
|
|
324
|
+
elif all(row.passed for row in scoreable):
|
|
325
|
+
if icc is not None and icc < unstable_icc_floor:
|
|
326
|
+
verdict = "unstable"
|
|
327
|
+
verdict_reason = "icc_below_floor"
|
|
328
|
+
else:
|
|
329
|
+
verdict = "pass"
|
|
330
|
+
elif all(not row.passed for row in scoreable):
|
|
331
|
+
verdict = "fail"
|
|
332
|
+
else:
|
|
333
|
+
verdict = "unstable"
|
|
334
|
+
verdict_reason = "mixed_outcomes"
|
|
335
|
+
|
|
336
|
+
if budget_exhausted and verdict == "pass":
|
|
337
|
+
# Hitting a cap mid-run yields `unstable` with reason budget_exhausted
|
|
338
|
+
# rather than a silently smaller n (ARCH §4 budget mechanics).
|
|
339
|
+
verdict = "unstable"
|
|
340
|
+
verdict_reason = "budget_exhausted"
|
|
341
|
+
|
|
342
|
+
if verdict == "unstable":
|
|
343
|
+
findings.append(
|
|
344
|
+
{
|
|
345
|
+
"type": "live_lane_scenario_unstable",
|
|
346
|
+
"level": "warning",
|
|
347
|
+
"detail": {
|
|
348
|
+
"reason": verdict_reason,
|
|
349
|
+
"icc": icc,
|
|
350
|
+
"divergence_step": fork_step,
|
|
351
|
+
},
|
|
352
|
+
}
|
|
353
|
+
)
|
|
354
|
+
|
|
355
|
+
return LaneRunResult(
|
|
356
|
+
lane=lane,
|
|
357
|
+
evidence_class=evidence_class,
|
|
358
|
+
repeats=repeats,
|
|
359
|
+
verdict=verdict,
|
|
360
|
+
per_repeat=[row.to_row() for row in rows],
|
|
361
|
+
icc=icc,
|
|
362
|
+
within_variance=within,
|
|
363
|
+
divergence_step=fork_step,
|
|
364
|
+
determinism=determinism,
|
|
365
|
+
quarantined_repeats=quarantined_count,
|
|
366
|
+
required_env=[str(name) for name in required_env],
|
|
367
|
+
end_state_diff=end_state_diff,
|
|
368
|
+
run_id=resolved_run_id,
|
|
369
|
+
rung=rung,
|
|
370
|
+
framework=framework_name,
|
|
371
|
+
framework_version=framework_version,
|
|
372
|
+
version_requirement=version_requirement,
|
|
373
|
+
version_ok=version_ok_observed,
|
|
374
|
+
repeats_requested=repeats,
|
|
375
|
+
repeats_completed=len(rows),
|
|
376
|
+
budget_cap_s=budget,
|
|
377
|
+
budget_spent_s=round(budget_spent, 6),
|
|
378
|
+
verdict_reason=verdict_reason,
|
|
379
|
+
findings=findings,
|
|
380
|
+
artifacts_dir=str(base_dir),
|
|
381
|
+
)
|
|
382
|
+
|
|
383
|
+
|
|
384
|
+
def primary_transcript_events(result: LaneRunResult) -> list[dict[str, Any]]:
|
|
385
|
+
"""Events of the first scoreable repeat (falling back to the first row) —
|
|
386
|
+
the transcript the lane normalizes into its state keys."""
|
|
387
|
+
|
|
388
|
+
from ._transcript import read_transcript
|
|
389
|
+
|
|
390
|
+
rows = [row for row in result.per_repeat if not row.get("quarantined")]
|
|
391
|
+
rows = rows or list(result.per_repeat)
|
|
392
|
+
for row in rows:
|
|
393
|
+
path = row.get("transcript_path")
|
|
394
|
+
if path and Path(str(path)).is_file():
|
|
395
|
+
return read_transcript(str(path))
|
|
396
|
+
return []
|
|
397
|
+
|
|
398
|
+
|
|
399
|
+
def lane_run_payload(
|
|
400
|
+
result: LaneRunResult,
|
|
401
|
+
*,
|
|
402
|
+
name: str | None = None,
|
|
403
|
+
scenario: Mapping[str, Any] | None = None,
|
|
404
|
+
manifest: Mapping[str, Any] | None = None,
|
|
405
|
+
states: Mapping[str, Any] | None = None,
|
|
406
|
+
metadata: Mapping[str, Any] | None = None,
|
|
407
|
+
) -> dict[str, Any]:
|
|
408
|
+
"""Serialize a lane run into the standard ``agent-learning.run.v1``
|
|
409
|
+
payload via the existing public envelope, with live-only fields under a
|
|
410
|
+
``live_lane`` evidence block — same artifact kind, same state keys, plus
|
|
411
|
+
live evidence (the graduation contract, R§3.1)."""
|
|
412
|
+
|
|
413
|
+
payload: dict[str, Any] = {
|
|
414
|
+
"kind": AGENT_LEARNING_RUN_KIND,
|
|
415
|
+
"name": str(name or f"live-{result.lane}-run-{result.run_id[:8]}"),
|
|
416
|
+
"evidence_class": result.evidence_class,
|
|
417
|
+
"live_lane": result.to_block(),
|
|
418
|
+
"findings": list(result.findings),
|
|
419
|
+
"summary": {
|
|
420
|
+
"verdict": result.verdict,
|
|
421
|
+
"verdict_reason": result.verdict_reason,
|
|
422
|
+
"repeats": result.repeats,
|
|
423
|
+
"repeats_completed": result.repeats_completed,
|
|
424
|
+
"quarantined_repeats": result.quarantined_repeats,
|
|
425
|
+
"icc": result.icc,
|
|
426
|
+
"divergence_step": result.divergence_step,
|
|
427
|
+
},
|
|
428
|
+
}
|
|
429
|
+
if scenario is not None:
|
|
430
|
+
payload["scenario"] = dict(scenario)
|
|
431
|
+
if manifest is not None:
|
|
432
|
+
payload["manifest"] = dict(manifest)
|
|
433
|
+
if states:
|
|
434
|
+
for state_key, state_value in states.items():
|
|
435
|
+
payload[str(state_key)] = state_value
|
|
436
|
+
if metadata:
|
|
437
|
+
payload["metadata"] = dict(metadata)
|
|
438
|
+
return public_payload(payload, kind=AGENT_LEARNING_RUN_KIND)
|
|
439
|
+
|
|
440
|
+
|
|
441
|
+
# --- dual-channel voice evidence (3B/3C — PRD §4.2 / guide §3.5) -------------
|
|
442
|
+
|
|
443
|
+
|
|
444
|
+
def _activity_mask(
|
|
445
|
+
pcm: np.ndarray,
|
|
446
|
+
*,
|
|
447
|
+
frame_samples: int,
|
|
448
|
+
energy_threshold_db: float,
|
|
449
|
+
) -> np.ndarray:
|
|
450
|
+
samples = np.asarray(pcm, dtype=float)
|
|
451
|
+
if samples.size == 0:
|
|
452
|
+
return np.zeros(0, dtype=bool)
|
|
453
|
+
peak = np.max(np.abs(samples))
|
|
454
|
+
if peak > 0:
|
|
455
|
+
samples = samples / peak
|
|
456
|
+
frame_count = int(np.ceil(samples.size / frame_samples))
|
|
457
|
+
padded = np.zeros(frame_count * frame_samples, dtype=float)
|
|
458
|
+
padded[: samples.size] = samples
|
|
459
|
+
frames = padded.reshape(frame_count, frame_samples)
|
|
460
|
+
rms = np.sqrt((frames**2).mean(axis=1))
|
|
461
|
+
with np.errstate(divide="ignore"):
|
|
462
|
+
rms_db = 20.0 * np.log10(np.where(rms > 0, rms, 1e-12))
|
|
463
|
+
return rms_db > energy_threshold_db
|
|
464
|
+
|
|
465
|
+
|
|
466
|
+
def _segments(mask: np.ndarray) -> list[tuple[int, int]]:
|
|
467
|
+
"""Contiguous active [start, end) frame spans."""
|
|
468
|
+
|
|
469
|
+
spans: list[tuple[int, int]] = []
|
|
470
|
+
start: int | None = None
|
|
471
|
+
for index, active in enumerate(mask):
|
|
472
|
+
if active and start is None:
|
|
473
|
+
start = index
|
|
474
|
+
elif not active and start is not None:
|
|
475
|
+
spans.append((start, index))
|
|
476
|
+
start = None
|
|
477
|
+
if start is not None:
|
|
478
|
+
spans.append((start, len(mask)))
|
|
479
|
+
return spans
|
|
480
|
+
|
|
481
|
+
|
|
482
|
+
def derive_channel_evidence(
|
|
483
|
+
user_pcm: np.ndarray,
|
|
484
|
+
agent_pcm: np.ndarray,
|
|
485
|
+
*,
|
|
486
|
+
sample_rate: int,
|
|
487
|
+
frame_ms: float = 20.0,
|
|
488
|
+
energy_threshold_db: float = -40.0,
|
|
489
|
+
) -> dict[str, Any]:
|
|
490
|
+
"""Compute the ``channels.derived`` block from the two PCM streams —
|
|
491
|
+
never from transcripts (R§3.5): barge-in latency, overlap totals,
|
|
492
|
+
post-interrupt recovery turns, and agent onset (ttfb). Pure-numpy
|
|
493
|
+
energy/onset detection; rung-2+ only (rung 1 has no channels block —
|
|
494
|
+
the rung-1 honesty rule, guide §3.5)."""
|
|
495
|
+
|
|
496
|
+
if sample_rate <= 0:
|
|
497
|
+
raise ValueError("sample_rate must be positive")
|
|
498
|
+
frame_samples = max(int(sample_rate * frame_ms / 1000.0), 1)
|
|
499
|
+
user_mask = _activity_mask(
|
|
500
|
+
user_pcm, frame_samples=frame_samples, energy_threshold_db=energy_threshold_db
|
|
501
|
+
)
|
|
502
|
+
agent_mask = _activity_mask(
|
|
503
|
+
agent_pcm, frame_samples=frame_samples, energy_threshold_db=energy_threshold_db
|
|
504
|
+
)
|
|
505
|
+
width = max(len(user_mask), len(agent_mask))
|
|
506
|
+
user_full = np.zeros(width, dtype=bool)
|
|
507
|
+
agent_full = np.zeros(width, dtype=bool)
|
|
508
|
+
user_full[: len(user_mask)] = user_mask
|
|
509
|
+
agent_full[: len(agent_mask)] = agent_mask
|
|
510
|
+
|
|
511
|
+
overlap = user_full & agent_full
|
|
512
|
+
overlap_spans = _segments(overlap)
|
|
513
|
+
overlap_total_ms = float(sum(end - start for start, end in overlap_spans)) * frame_ms
|
|
514
|
+
|
|
515
|
+
user_spans = _segments(user_full)
|
|
516
|
+
agent_spans = _segments(agent_full)
|
|
517
|
+
|
|
518
|
+
# ttfb: first agent onset after the first user utterance ends.
|
|
519
|
+
ttfb_ms: float | None = None
|
|
520
|
+
if user_spans and agent_spans:
|
|
521
|
+
first_user_end = user_spans[0][1]
|
|
522
|
+
for start, _ in agent_spans:
|
|
523
|
+
if start >= first_user_end:
|
|
524
|
+
ttfb_ms = float(start - first_user_end) * frame_ms
|
|
525
|
+
break
|
|
526
|
+
|
|
527
|
+
# barge-in: first user onset that lands mid-agent-speech; latency runs
|
|
528
|
+
# until the agent yields (its active span ends).
|
|
529
|
+
barge_in_latency_ms: float | None = None
|
|
530
|
+
barge_frame: int | None = None
|
|
531
|
+
for user_start, _ in user_spans:
|
|
532
|
+
for agent_start, agent_end in agent_spans:
|
|
533
|
+
if agent_start < user_start < agent_end:
|
|
534
|
+
barge_in_latency_ms = float(agent_end - user_start) * frame_ms
|
|
535
|
+
barge_frame = user_start
|
|
536
|
+
break
|
|
537
|
+
if barge_in_latency_ms is not None:
|
|
538
|
+
break
|
|
539
|
+
|
|
540
|
+
# recovery: agent speech segments after the interrupt until the first
|
|
541
|
+
# segment that starts clear of user speech (a clean turn).
|
|
542
|
+
post_interrupt_recovery_turns: int | None = None
|
|
543
|
+
if barge_frame is not None:
|
|
544
|
+
turns = 0
|
|
545
|
+
for agent_start, _ in agent_spans:
|
|
546
|
+
if agent_start <= barge_frame:
|
|
547
|
+
continue
|
|
548
|
+
turns += 1
|
|
549
|
+
if not user_full[agent_start]:
|
|
550
|
+
break
|
|
551
|
+
post_interrupt_recovery_turns = turns
|
|
552
|
+
|
|
553
|
+
return {
|
|
554
|
+
"barge_in_latency_ms": barge_in_latency_ms,
|
|
555
|
+
"overlap_total_ms": overlap_total_ms,
|
|
556
|
+
"overlap_segments": len(overlap_spans),
|
|
557
|
+
"post_interrupt_recovery_turns": post_interrupt_recovery_turns,
|
|
558
|
+
"ttfb_ms": ttfb_ms,
|
|
559
|
+
"frame_ms": frame_ms,
|
|
560
|
+
"energy_threshold_db": energy_threshold_db,
|
|
561
|
+
}
|