agent-learning-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
- agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
- agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
- agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
- agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
- agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
- fi/__init__.py +5 -0
- fi/alk/__init__.py +57 -0
- fi/alk/_facade.py +31 -0
- fi/alk/_module_alias.py +68 -0
- fi/alk/_paths.py +14 -0
- fi/alk/_schema.py +522 -0
- fi/alk/actions.py +727 -0
- fi/alk/bench/__init__.py +517 -0
- fi/alk/bench/_codeexec.py +213 -0
- fi/alk/bench/_coding.py +215 -0
- fi/alk/bench/_docker.py +237 -0
- fi/alk/bench/_grader.py +286 -0
- fi/alk/bench/_pull.py +212 -0
- fi/alk/bench/_voice.py +147 -0
- fi/alk/capabilities.py +627 -0
- fi/alk/cli.py +6396 -0
- fi/alk/config.py +130 -0
- fi/alk/cua_loop.py +562 -0
- fi/alk/evals.py +2351 -0
- fi/alk/extensions.py +163 -0
- fi/alk/harness/ARCHITECTURE.md +231 -0
- fi/alk/harness/DESIGN.md +246 -0
- fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
- fi/alk/harness/HOW-IT-WORKS.md +297 -0
- fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
- fi/alk/harness/README.md +417 -0
- fi/alk/harness/__init__.py +77 -0
- fi/alk/harness/__main__.py +3 -0
- fi/alk/harness/amend.py +312 -0
- fi/alk/harness/artifacts.py +319 -0
- fi/alk/harness/authoring_entrypoint.py +189 -0
- fi/alk/harness/authoring_runtime_validation.py +267 -0
- fi/alk/harness/backends/README.md +43 -0
- fi/alk/harness/backends/__init__.py +122 -0
- fi/alk/harness/backends/base.py +241 -0
- fi/alk/harness/backends/claude.py +211 -0
- fi/alk/harness/backends/files.py +182 -0
- fi/alk/harness/backends/vertex_gemini.py +457 -0
- fi/alk/harness/background_noise.py +95 -0
- fi/alk/harness/build.py +385 -0
- fi/alk/harness/bundle.py +593 -0
- fi/alk/harness/bundle_author_v2.py +1831 -0
- fi/alk/harness/bundle_v2.py +719 -0
- fi/alk/harness/call_runner.py +1440 -0
- fi/alk/harness/callback_http_adapter.py +111 -0
- fi/alk/harness/catalogue.py +287 -0
- fi/alk/harness/chat.py +428 -0
- fi/alk/harness/chat_call_runner.py +506 -0
- fi/alk/harness/checks.py +136 -0
- fi/alk/harness/cli.py +1354 -0
- fi/alk/harness/config.py +338 -0
- fi/alk/harness/contract.py +718 -0
- fi/alk/harness/credentials.py +674 -0
- fi/alk/harness/data/persona_vocabulary.json +111 -0
- fi/alk/harness/environment.py +99 -0
- fi/alk/harness/environment_plan.py +168 -0
- fi/alk/harness/events.py +125 -0
- fi/alk/harness/executor.py +304 -0
- fi/alk/harness/folder.py +234 -0
- fi/alk/harness/generated_runtime.py +815 -0
- fi/alk/harness/github.py +72 -0
- fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
- fi/alk/harness/hosted_entrypoint.py +2402 -0
- fi/alk/harness/hosted_scheduler.py +2218 -0
- fi/alk/harness/job.py +426 -0
- fi/alk/harness/judge.py +184 -0
- fi/alk/harness/livekit_source.py +50 -0
- fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
- fi/alk/harness/observability.py +208 -0
- fi/alk/harness/outbound.py +3252 -0
- fi/alk/harness/packaging.py +515 -0
- fi/alk/harness/persona_guides.py +157 -0
- fi/alk/harness/platform.py +692 -0
- fi/alk/harness/process_preflight.py +764 -0
- fi/alk/harness/process_runtime.py +5670 -0
- fi/alk/harness/prove.py +425 -0
- fi/alk/harness/provider_import.py +703 -0
- fi/alk/harness/provider_lifecycle.py +392 -0
- fi/alk/harness/provision.py +2896 -0
- fi/alk/harness/reception.py +147 -0
- fi/alk/harness/retell_chat_call_runner.py +373 -0
- fi/alk/harness/run/__init__.py +296 -0
- fi/alk/harness/run/alk.py +184 -0
- fi/alk/harness/run/call.py +162 -0
- fi/alk/harness/run/conversation.py +264 -0
- fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
- fi/alk/harness/run/evidence.py +195 -0
- fi/alk/harness/run/grade.py +598 -0
- fi/alk/harness/run/live.py +297 -0
- fi/alk/harness/run/models.py +56 -0
- fi/alk/harness/run/platform_evals.py +227 -0
- fi/alk/harness/run/sdk_voice.py +130 -0
- fi/alk/harness/run/simulation.py +1209 -0
- fi/alk/harness/run/stage.py +91 -0
- fi/alk/harness/run/targets.py +508 -0
- fi/alk/harness/run/tools.py +601 -0
- fi/alk/harness/run/voice.py +340 -0
- fi/alk/harness/runtime.py +172 -0
- fi/alk/harness/sandbox_server.py +2011 -0
- fi/alk/harness/sandbox_worker.py +44 -0
- fi/alk/harness/scenario.py +1048 -0
- fi/alk/harness/scenario_source.py +879 -0
- fi/alk/harness/scenario_tools.py +1143 -0
- fi/alk/harness/scenarios.py +915 -0
- fi/alk/harness/secrets.py +168 -0
- fi/alk/harness/service_catalog.py +97 -0
- fi/alk/harness/session.py +391 -0
- fi/alk/harness/sessions.py +372 -0
- fi/alk/harness/simulator.py +76 -0
- fi/alk/harness/simulator_voice.py +928 -0
- fi/alk/harness/skills/build-environment/SKILL.md +538 -0
- fi/alk/harness/skills/harness.md +131 -0
- fi/alk/harness/skills/kinds/chat.md +48 -0
- fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
- fi/alk/harness/skills/kinds/voice.md +59 -0
- fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
- fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
- fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
- fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
- fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
- fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
- fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
- fi/alk/harness/source_data_invariants.py +444 -0
- fi/alk/harness/source_tool_evidence.py +79 -0
- fi/alk/harness/sources.py +253 -0
- fi/alk/harness/spend.py +140 -0
- fi/alk/harness/tool_trace_proxy.py +104 -0
- fi/alk/harness/tools.py +1018 -0
- fi/alk/harness/understand.py +169 -0
- fi/alk/harness/voicemail_audio.py +74 -0
- fi/alk/harness/world/__init__.py +33 -0
- fi/alk/harness/world/errors.py +68 -0
- fi/alk/harness/world/expectations.py +91 -0
- fi/alk/harness/world/handle.py +538 -0
- fi/alk/harness/world/kinds.py +196 -0
- fi/alk/harness/world/mutate.py +186 -0
- fi/alk/harness/world/probe.py +413 -0
- fi/alk/harness/world/provision.py +511 -0
- fi/alk/harness/world/provisioned.py +191 -0
- fi/alk/harness/world/runtime.py +616 -0
- fi/alk/harness/world/snapshot.py +288 -0
- fi/alk/harness/world/stores/__init__.py +305 -0
- fi/alk/harness/world/stores/container.py +215 -0
- fi/alk/harness/world/stores/inprocess.py +346 -0
- fi/alk/harness/world/stores/postgres.py +481 -0
- fi/alk/harness/world/stores/prove.py +202 -0
- fi/alk/harness/world/stores/sqlite.py +245 -0
- fi/alk/harness/world/stores/written.py +182 -0
- fi/alk/harness/world/tools.py +1516 -0
- fi/alk/harness/world/workspace.py +144 -0
- fi/alk/image_loop.py +453 -0
- fi/alk/image_perturb.py +241 -0
- fi/alk/improve.py +274 -0
- fi/alk/live/__init__.py +154 -0
- fi/alk/live/_attribution.py +184 -0
- fi/alk/live/_capture.py +264 -0
- fi/alk/live/_codec.py +391 -0
- fi/alk/live/_contract.py +134 -0
- fi/alk/live/_loopback.py +316 -0
- fi/alk/live/_perturb.py +449 -0
- fi/alk/live/_runner.py +386 -0
- fi/alk/live/_stats.py +561 -0
- fi/alk/live/_transcript.py +240 -0
- fi/alk/live/_workers/__init__.py +9 -0
- fi/alk/live/_workers/a2a_worker.py +316 -0
- fi/alk/live/_workers/langgraph_worker.py +217 -0
- fi/alk/live/_workers/livekit_worker.py +207 -0
- fi/alk/live/_workers/mcp_loopback_server.py +46 -0
- fi/alk/live/_workers/mcp_worker.py +158 -0
- fi/alk/live/_workers/pipecat_worker.py +189 -0
- fi/alk/live/a2a_lane.py +138 -0
- fi/alk/live/langgraph_lane.py +339 -0
- fi/alk/live/livekit_lane.py +376 -0
- fi/alk/live/mcp_lane.py +172 -0
- fi/alk/live/pipecat_lane.py +341 -0
- fi/alk/live/voice_redteam.py +494 -0
- fi/alk/loss.py +306 -0
- fi/alk/optimize.py +36260 -0
- fi/alk/practice/__init__.py +51 -0
- fi/alk/practice/_assess.py +103 -0
- fi/alk/practice/_budget.py +81 -0
- fi/alk/practice/_calibrate.py +69 -0
- fi/alk/practice/_capstone.py +86 -0
- fi/alk/practice/_contract.py +91 -0
- fi/alk/practice/_diagnose.py +79 -0
- fi/alk/practice/_drill.py +196 -0
- fi/alk/practice/_experiment.py +720 -0
- fi/alk/practice/_schedule.py +102 -0
- fi/alk/practice/_store.py +194 -0
- fi/alk/practice/_trainer.py +245 -0
- fi/alk/practice/_update.py +125 -0
- fi/alk/redteam.py +2621 -0
- fi/alk/rewardhack.py +237 -0
- fi/alk/simulate.py +10351 -0
- fi/alk/studio/__init__.py +82 -0
- fi/alk/studio/_bias.py +314 -0
- fi/alk/studio/_calibration.py +522 -0
- fi/alk/studio/_coverage.py +262 -0
- fi/alk/studio/_download.py +665 -0
- fi/alk/studio/_fidelity_attack.py +114 -0
- fi/alk/studio/_generate.py +652 -0
- fi/alk/studio/_library.py +370 -0
- fi/alk/studio/_scan.py +134 -0
- fi/alk/studio/_upgrade.py +42 -0
- fi/alk/studio/_vendor.py +172 -0
- fi/alk/suite.py +4200 -0
- fi/alk/tasks.py +828 -0
- fi/alk/telemetry/__init__.py +149 -0
- fi/alk/telemetry/_contract.py +141 -0
- fi/alk/telemetry/_emit.py +182 -0
- fi/alk/telemetry/_ledger.py +296 -0
- fi/alk/telemetry/_queue.py +127 -0
- fi/alk/telemetry/_row.py +294 -0
- fi/alk/telemetry/_run.py +233 -0
- fi/alk/telemetry/_sync.py +193 -0
- fi/alk/telemetry/_url.py +119 -0
- fi/alk/trinity.py +49397 -0
- fi/alk/voice_loop.py +174 -0
- fi/api/__init__.py +1 -0
- fi/api/auth.py +137 -0
- fi/api/types.py +29 -0
- fi/cli/__init__.py +9 -0
- fi/cli/assertions/__init__.py +25 -0
- fi/cli/assertions/conditions.py +76 -0
- fi/cli/assertions/evaluator.py +286 -0
- fi/cli/assertions/exit_codes.py +20 -0
- fi/cli/assertions/parser.py +131 -0
- fi/cli/assertions/reporter.py +194 -0
- fi/cli/commands/__init__.py +9 -0
- fi/cli/commands/config.py +165 -0
- fi/cli/commands/export.py +208 -0
- fi/cli/commands/init.py +112 -0
- fi/cli/commands/list_cmd.py +213 -0
- fi/cli/commands/run.py +486 -0
- fi/cli/commands/validate.py +173 -0
- fi/cli/commands/view.py +424 -0
- fi/cli/config/__init__.py +6 -0
- fi/cli/config/defaults.py +206 -0
- fi/cli/config/loader.py +155 -0
- fi/cli/config/schema.py +174 -0
- fi/cli/main.py +78 -0
- fi/cli/output/__init__.py +6 -0
- fi/cli/output/formatters.py +106 -0
- fi/cli/output/reporters.py +46 -0
- fi/cli/storage/__init__.py +5 -0
- fi/cli/storage/run_history.py +249 -0
- fi/cli/utils/__init__.py +5 -0
- fi/cli/utils/console.py +44 -0
- fi/evals/__init__.py +131 -0
- fi/evals/autoeval/__init__.py +137 -0
- fi/evals/autoeval/analyzer.py +211 -0
- fi/evals/autoeval/config.py +244 -0
- fi/evals/autoeval/export.py +213 -0
- fi/evals/autoeval/interactive.py +283 -0
- fi/evals/autoeval/pipeline.py +625 -0
- fi/evals/autoeval/prompts.py +139 -0
- fi/evals/autoeval/recommender.py +242 -0
- fi/evals/autoeval/rules.py +589 -0
- fi/evals/autoeval/templates.py +299 -0
- fi/evals/autoeval/types.py +232 -0
- fi/evals/core/__init__.py +16 -0
- fi/evals/core/cloud_registry.py +184 -0
- fi/evals/core/engines.py +368 -0
- fi/evals/core/evaluate.py +319 -0
- fi/evals/core/judge_prompt.py +90 -0
- fi/evals/core/prompt_generator.py +83 -0
- fi/evals/core/registry.py +57 -0
- fi/evals/core/result.py +55 -0
- fi/evals/evaluator.py +721 -0
- fi/evals/execution.py +168 -0
- fi/evals/feedback/__init__.py +32 -0
- fi/evals/feedback/calibrator.py +160 -0
- fi/evals/feedback/collector.py +214 -0
- fi/evals/feedback/hooks.py +81 -0
- fi/evals/feedback/retriever.py +128 -0
- fi/evals/feedback/store.py +272 -0
- fi/evals/feedback/types.py +99 -0
- fi/evals/framework/README.md +79 -0
- fi/evals/framework/__init__.py +267 -0
- fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
- fi/evals/framework/backends/__init__.py +99 -0
- fi/evals/framework/backends/_container.py +141 -0
- fi/evals/framework/backends/_utils.py +145 -0
- fi/evals/framework/backends/base.py +223 -0
- fi/evals/framework/backends/celery_backend.py +417 -0
- fi/evals/framework/backends/celery_worker.py +78 -0
- fi/evals/framework/backends/kubernetes_backend.py +665 -0
- fi/evals/framework/backends/ray_backend.py +521 -0
- fi/evals/framework/backends/temporal.py +350 -0
- fi/evals/framework/backends/temporal_worker.py +126 -0
- fi/evals/framework/backends/thread_pool.py +286 -0
- fi/evals/framework/context.py +258 -0
- fi/evals/framework/enrichment.py +306 -0
- fi/evals/framework/evals/__init__.py +68 -0
- fi/evals/framework/evals/agentic.py +399 -0
- fi/evals/framework/evals/builder.py +609 -0
- fi/evals/framework/evals/semantic.py +142 -0
- fi/evals/framework/evaluator.py +647 -0
- fi/evals/framework/evaluators/__init__.py +22 -0
- fi/evals/framework/evaluators/blocking.py +347 -0
- fi/evals/framework/evaluators/non_blocking.py +577 -0
- fi/evals/framework/propagation.py +421 -0
- fi/evals/framework/protocols.py +385 -0
- fi/evals/framework/registry.py +370 -0
- fi/evals/framework/resilience/__init__.py +150 -0
- fi/evals/framework/resilience/circuit_breaker.py +309 -0
- fi/evals/framework/resilience/degradation.py +355 -0
- fi/evals/framework/resilience/health.py +505 -0
- fi/evals/framework/resilience/rate_limiter.py +228 -0
- fi/evals/framework/resilience/retry.py +274 -0
- fi/evals/framework/resilience/types.py +288 -0
- fi/evals/framework/resilience/wrapper.py +433 -0
- fi/evals/framework/types.py +218 -0
- fi/evals/guardrails/README.md +915 -0
- fi/evals/guardrails/__init__.py +96 -0
- fi/evals/guardrails/backends/__init__.py +43 -0
- fi/evals/guardrails/backends/azure.py +361 -0
- fi/evals/guardrails/backends/base.py +88 -0
- fi/evals/guardrails/backends/generic_llm.py +163 -0
- fi/evals/guardrails/backends/granite.py +216 -0
- fi/evals/guardrails/backends/llamaguard.py +221 -0
- fi/evals/guardrails/backends/local_base.py +479 -0
- fi/evals/guardrails/backends/openai.py +365 -0
- fi/evals/guardrails/backends/qwen.py +170 -0
- fi/evals/guardrails/backends/shieldgemma.py +154 -0
- fi/evals/guardrails/backends/turing.py +235 -0
- fi/evals/guardrails/backends/vllm_client.py +321 -0
- fi/evals/guardrails/backends/wildguard.py +188 -0
- fi/evals/guardrails/base.py +888 -0
- fi/evals/guardrails/config.py +221 -0
- fi/evals/guardrails/discovery.py +243 -0
- fi/evals/guardrails/gateway.py +437 -0
- fi/evals/guardrails/registry.py +231 -0
- fi/evals/guardrails/scanners/__init__.py +127 -0
- fi/evals/guardrails/scanners/base.py +191 -0
- fi/evals/guardrails/scanners/code_injection.py +243 -0
- fi/evals/guardrails/scanners/eval_delegate.py +574 -0
- fi/evals/guardrails/scanners/invisible_chars.py +351 -0
- fi/evals/guardrails/scanners/jailbreak.py +412 -0
- fi/evals/guardrails/scanners/language.py +288 -0
- fi/evals/guardrails/scanners/pipeline.py +260 -0
- fi/evals/guardrails/scanners/regex.py +311 -0
- fi/evals/guardrails/scanners/secrets.py +274 -0
- fi/evals/guardrails/scanners/topics.py +649 -0
- fi/evals/guardrails/scanners/urls.py +341 -0
- fi/evals/guardrails/types.py +96 -0
- fi/evals/llm/__init__.py +3 -0
- fi/evals/llm/base_llm_provider.py +35 -0
- fi/evals/llm/providers/litellm.py +70 -0
- fi/evals/local/__init__.py +90 -0
- fi/evals/local/evaluator.py +690 -0
- fi/evals/local/execution_mode.py +121 -0
- fi/evals/local/llm.py +489 -0
- fi/evals/local/metrics/__init__.py +19 -0
- fi/evals/local/registry.py +360 -0
- fi/evals/manager.py +1018 -0
- fi/evals/manager_types.py +362 -0
- fi/evals/metrics/__init__.py +185 -0
- fi/evals/metrics/agents/__init__.py +74 -0
- fi/evals/metrics/agents/metrics.py +693 -0
- fi/evals/metrics/agents/report.py +36463 -0
- fi/evals/metrics/agents/types.py +160 -0
- fi/evals/metrics/base_llm_metric.py +111 -0
- fi/evals/metrics/base_metric.py +138 -0
- fi/evals/metrics/code_security/__init__.py +305 -0
- fi/evals/metrics/code_security/analyzer.py +985 -0
- fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
- fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
- fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
- fi/evals/metrics/code_security/benchmarks/types.py +308 -0
- fi/evals/metrics/code_security/detectors/__init__.py +186 -0
- fi/evals/metrics/code_security/detectors/base.py +394 -0
- fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
- fi/evals/metrics/code_security/detectors/injection.py +744 -0
- fi/evals/metrics/code_security/detectors/secrets.py +287 -0
- fi/evals/metrics/code_security/detectors/serialization.py +192 -0
- fi/evals/metrics/code_security/joint_metrics.py +588 -0
- fi/evals/metrics/code_security/judges/__init__.py +83 -0
- fi/evals/metrics/code_security/judges/base.py +238 -0
- fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
- fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
- fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
- fi/evals/metrics/code_security/metrics.py +388 -0
- fi/evals/metrics/code_security/modes/__init__.py +63 -0
- fi/evals/metrics/code_security/modes/adversarial.py +284 -0
- fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
- fi/evals/metrics/code_security/modes/base.py +283 -0
- fi/evals/metrics/code_security/modes/instruct.py +253 -0
- fi/evals/metrics/code_security/modes/repair.py +230 -0
- fi/evals/metrics/code_security/reports/__init__.py +57 -0
- fi/evals/metrics/code_security/reports/generator.py +404 -0
- fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
- fi/evals/metrics/code_security/types.py +534 -0
- fi/evals/metrics/function_calling/__init__.py +34 -0
- fi/evals/metrics/function_calling/metrics.py +573 -0
- fi/evals/metrics/function_calling/types.py +87 -0
- fi/evals/metrics/hallucination/__init__.py +54 -0
- fi/evals/metrics/hallucination/detector.py +149 -0
- fi/evals/metrics/hallucination/metrics.py +390 -0
- fi/evals/metrics/hallucination/nli.py +253 -0
- fi/evals/metrics/hallucination/sentinel.py +106 -0
- fi/evals/metrics/hallucination/types.py +132 -0
- fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
- fi/evals/metrics/heuristics/json_metrics.py +87 -0
- fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
- fi/evals/metrics/heuristics/string_metrics.py +391 -0
- fi/evals/metrics/llm_as_judges/__init__.py +17 -0
- fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
- fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
- fi/evals/metrics/llm_as_judges/types.py +48 -0
- fi/evals/metrics/rag/__init__.py +111 -0
- fi/evals/metrics/rag/advanced/__init__.py +14 -0
- fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
- fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
- fi/evals/metrics/rag/generation/__init__.py +17 -0
- fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
- fi/evals/metrics/rag/generation/context_utilization.py +245 -0
- fi/evals/metrics/rag/generation/faithfulness.py +241 -0
- fi/evals/metrics/rag/generation/groundedness.py +131 -0
- fi/evals/metrics/rag/rag_score.py +277 -0
- fi/evals/metrics/rag/retrieval/__init__.py +20 -0
- fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
- fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
- fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
- fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
- fi/evals/metrics/rag/retrieval/ranking.py +261 -0
- fi/evals/metrics/rag/types.py +100 -0
- fi/evals/metrics/rag/utils/__init__.py +62 -0
- fi/evals/metrics/rag/utils/claims.py +189 -0
- fi/evals/metrics/rag/utils/entities.py +244 -0
- fi/evals/metrics/rag/utils/nli.py +92 -0
- fi/evals/metrics/rag/utils/similarity.py +345 -0
- fi/evals/metrics/structured/__init__.py +114 -0
- fi/evals/metrics/structured/field_completeness.py +313 -0
- fi/evals/metrics/structured/hierarchy_score.py +366 -0
- fi/evals/metrics/structured/json_validation.py +190 -0
- fi/evals/metrics/structured/schema_compliance.py +280 -0
- fi/evals/metrics/structured/structured_output_score.py +298 -0
- fi/evals/metrics/structured/types.py +108 -0
- fi/evals/metrics/structured/validators/__init__.py +30 -0
- fi/evals/metrics/structured/validators/base.py +189 -0
- fi/evals/metrics/structured/validators/json_validator.py +196 -0
- fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
- fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
- fi/evals/otel/__init__.py +266 -0
- fi/evals/otel/config.py +400 -0
- fi/evals/otel/conventions.py +463 -0
- fi/evals/otel/enrichment.py +371 -0
- fi/evals/otel/instrumentors/__init__.py +140 -0
- fi/evals/otel/instrumentors/anthropic.py +517 -0
- fi/evals/otel/instrumentors/base.py +382 -0
- fi/evals/otel/instrumentors/openai.py +673 -0
- fi/evals/otel/processors/__init__.py +36 -0
- fi/evals/otel/processors/base.py +473 -0
- fi/evals/otel/processors/cost.py +445 -0
- fi/evals/otel/processors/evaluation.py +559 -0
- fi/evals/otel/processors/llm.py +462 -0
- fi/evals/otel/tracer.py +506 -0
- fi/evals/otel/types.py +232 -0
- fi/evals/otel_utils.py +23 -0
- fi/evals/protect.py +671 -0
- fi/evals/protect_input_adapter.py +154 -0
- fi/evals/streaming/__init__.py +88 -0
- fi/evals/streaming/buffer.py +213 -0
- fi/evals/streaming/evaluator.py +551 -0
- fi/evals/streaming/policy.py +307 -0
- fi/evals/streaming/scorers.py +368 -0
- fi/evals/streaming/types.py +238 -0
- fi/evals/templates.py +472 -0
- fi/evals/types.py +156 -0
- fi/opt/__init__.py +221 -0
- fi/opt/_objective_scoring.py +85 -0
- fi/opt/base/__init__.py +11 -0
- fi/opt/base/base_generator.py +33 -0
- fi/opt/base/base_mapper.py +26 -0
- fi/opt/base/base_optimizer.py +45 -0
- fi/opt/base/evaluator.py +211 -0
- fi/opt/components.py +3095 -0
- fi/opt/datamappers/__init__.py +3 -0
- fi/opt/datamappers/basic_mapper.py +40 -0
- fi/opt/deployment.py +1021 -0
- fi/opt/evidence.py +4332 -0
- fi/opt/generators/__init__.py +3 -0
- fi/opt/generators/litellm.py +66 -0
- fi/opt/integrations/__init__.py +23 -0
- fi/opt/integrations/generative_suite.py +410 -0
- fi/opt/integrations/simulate.py +1313 -0
- fi/opt/mutations.py +771 -0
- fi/opt/observability.py +4639 -0
- fi/opt/optimizer_trace.py +889 -0
- fi/opt/optimizers/__init__.py +80 -0
- fi/opt/optimizers/agent.py +331 -0
- fi/opt/optimizers/agent_bandit.py +392 -0
- fi/opt/optimizers/agent_curriculum.py +635 -0
- fi/opt/optimizers/agent_evolution.py +894 -0
- fi/opt/optimizers/agent_feedback.py +1863 -0
- fi/opt/optimizers/agent_pareto.py +547 -0
- fi/opt/optimizers/agent_social_memory.py +1113 -0
- fi/opt/optimizers/agent_tpe.py +321 -0
- fi/opt/optimizers/bayesian_search.py +449 -0
- fi/opt/optimizers/council.py +2075 -0
- fi/opt/optimizers/futureagi_replay.py +799 -0
- fi/opt/optimizers/gepa.py +322 -0
- fi/opt/optimizers/metaprompt.py +243 -0
- fi/opt/optimizers/promptwizard.py +417 -0
- fi/opt/optimizers/protegi.py +329 -0
- fi/opt/optimizers/random_search.py +224 -0
- fi/opt/research.py +518 -0
- fi/opt/simulation.py +260 -0
- fi/opt/targets.py +232 -0
- fi/opt/types.py +66 -0
- fi/opt/utils/__init__.py +4 -0
- fi/opt/utils/early_stopping.py +266 -0
- fi/opt/utils/setup_logging.py +82 -0
- fi/simulate/__init__.py +540 -0
- fi/simulate/_hashing.py +35 -0
- fi/simulate/_logging.py +10 -0
- fi/simulate/adapters.py +87 -0
- fi/simulate/agent/__init__.py +120 -0
- fi/simulate/agent/browser.py +658 -0
- fi/simulate/agent/definition.py +587 -0
- fi/simulate/agent/frameworks.py +3528 -0
- fi/simulate/agent/generic.py +8286 -0
- fi/simulate/agent/import_probe.py +227 -0
- fi/simulate/agent/memory.py +905 -0
- fi/simulate/agent/mocks.py +101 -0
- fi/simulate/agent/multi_agent.py +361 -0
- fi/simulate/agent/orchestration.py +903 -0
- fi/simulate/agent/realtime.py +665 -0
- fi/simulate/agent/wrapper.py +99 -0
- fi/simulate/agent/wrappers/__init__.py +18 -0
- fi/simulate/agent/wrappers/anthropic.py +62 -0
- fi/simulate/agent/wrappers/gemini.py +65 -0
- fi/simulate/agent/wrappers/http.py +404 -0
- fi/simulate/agent/wrappers/langchain.py +80 -0
- fi/simulate/agent/wrappers/openai.py +75 -0
- fi/simulate/agent/wrappers/websocket.py +326 -0
- fi/simulate/artifacts/__init__.py +11 -0
- fi/simulate/artifacts/manifest.py +62 -0
- fi/simulate/cli.py +20560 -0
- fi/simulate/endpoints/__init__.py +45 -0
- fi/simulate/endpoints/_http_actor.py +73 -0
- fi/simulate/endpoints/actor_sources.py +243 -0
- fi/simulate/endpoints/base.py +107 -0
- fi/simulate/endpoints/builtins.py +10 -0
- fi/simulate/endpoints/callable.py +95 -0
- fi/simulate/endpoints/http.py +76 -0
- fi/simulate/endpoints/livekit.py +138 -0
- fi/simulate/endpoints/originators.py +132 -0
- fi/simulate/endpoints/profiles.py +348 -0
- fi/simulate/endpoints/retell.py +633 -0
- fi/simulate/endpoints/vapi.py +205 -0
- fi/simulate/endpoints/websocket.py +76 -0
- fi/simulate/environment.py +33026 -0
- fi/simulate/environments/__init__.py +11 -0
- fi/simulate/environments/base.py +73 -0
- fi/simulate/environments/chat.py +697 -0
- fi/simulate/environments/voice.py +212 -0
- fi/simulate/evaluation/__init__.py +4 -0
- fi/simulate/evaluation/ai_eval.py +227 -0
- fi/simulate/evidence/__init__.py +35 -0
- fi/simulate/evidence/base.py +59 -0
- fi/simulate/evidence/caller_observed.py +50 -0
- fi/simulate/evidence/livekit_instrumentation.py +51 -0
- fi/simulate/evidence/livekit_room.py +50 -0
- fi/simulate/evidence/otel.py +49 -0
- fi/simulate/evidence/providers/__init__.py +24 -0
- fi/simulate/evidence/providers/base.py +61 -0
- fi/simulate/evidence/providers/retell.py +376 -0
- fi/simulate/evidence/providers/vapi.py +426 -0
- fi/simulate/hosted/__init__.py +32 -0
- fi/simulate/hosted/child_entrypoint.py +306 -0
- fi/simulate/hosted/job.py +150 -0
- fi/simulate/hosted/targets.py +53 -0
- fi/simulate/instrumentation/__init__.py +5 -0
- fi/simulate/instrumentation/livekit/__init__.py +122 -0
- fi/simulate/manifest.py +1033 -0
- fi/simulate/matrix_cli.py +165 -0
- fi/simulate/realtime/__init__.py +40 -0
- fi/simulate/realtime/events.py +107 -0
- fi/simulate/realtime/media.py +61 -0
- fi/simulate/realtime/session.py +91 -0
- fi/simulate/recording/__init__.py +5 -0
- fi/simulate/recording/room_recorder.py +326 -0
- fi/simulate/registry.py +185 -0
- fi/simulate/results/__init__.py +9 -0
- fi/simulate/results/base.py +18 -0
- fi/simulate/results/filesystem.py +71 -0
- fi/simulate/results/futureagi.py +1340 -0
- fi/simulate/runtime/__init__.py +85 -0
- fi/simulate/runtime/capabilities.py +40 -0
- fi/simulate/runtime/events.py +63 -0
- fi/simulate/runtime/failures.py +25 -0
- fi/simulate/runtime/ids.py +34 -0
- fi/simulate/runtime/plan.py +70 -0
- fi/simulate/runtime/planner.py +102 -0
- fi/simulate/runtime/report.py +174 -0
- fi/simulate/runtime/run.py +75 -0
- fi/simulate/runtime/runner.py +333 -0
- fi/simulate/runtime/spec.py +186 -0
- fi/simulate/simulation/__init__.py +30 -0
- fi/simulate/simulation/behavior_policy.py +425 -0
- fi/simulate/simulation/bridge/__init__.py +9 -0
- fi/simulate/simulation/bridge/audio.py +29 -0
- fi/simulate/simulation/bridge/connector.py +46 -0
- fi/simulate/simulation/bridge/livekit.py +252 -0
- fi/simulate/simulation/bridge/retell.py +188 -0
- fi/simulate/simulation/bridge/vapi.py +177 -0
- fi/simulate/simulation/contract.py +419 -0
- fi/simulate/simulation/engines/__init__.py +12 -0
- fi/simulate/simulation/engines/base.py +21 -0
- fi/simulate/simulation/engines/cloud.py +517 -0
- fi/simulate/simulation/engines/livekit.py +4167 -0
- fi/simulate/simulation/engines/local_text.py +89 -0
- fi/simulate/simulation/fidelity.py +374 -0
- fi/simulate/simulation/gemini_tts_stream.py +110 -0
- fi/simulate/simulation/generator.py +91 -0
- fi/simulate/simulation/goal_machine.py +185 -0
- fi/simulate/simulation/livekit_models.py +467 -0
- fi/simulate/simulation/matrix.py +170 -0
- fi/simulate/simulation/models.py +279 -0
- fi/simulate/simulation/runner.py +153 -0
- fi/simulate/simulation/synthetic.py +880 -0
- fi/simulate/simulation/voice_prompt.py +502 -0
- fi/simulate/simulator/__init__.py +55 -0
- fi/simulate/simulator/builtins.py +53 -0
- fi/simulate/suite.py +1288 -0
- fi/simulate/utils/routes.py +164 -0
- fi/simulate/voice.py +225 -0
- fi/simulate/voice_cli.py +182 -0
- fi/utils/__init__.py +1 -0
- fi/utils/constants.py +14 -0
- fi/utils/errors.py +200 -0
- fi/utils/executor.py +26 -0
- fi/utils/routes.py +119 -0
- fi/utils/utils.py +17 -0
|
@@ -0,0 +1,425 @@
|
|
|
1
|
+
"""Behavior-policy compiler + per-axis realization metrics (Phase 7, unit 2).
|
|
2
|
+
|
|
3
|
+
Engine-side home (ARCH Decision 3): stdlib only — deterministic, no LLM, no
|
|
4
|
+
numpy. The six policy parameters map 1:1 onto the canon behavior axes, each
|
|
5
|
+
paired with its transcript-observable realization metric; a parameter without
|
|
6
|
+
one DOES NOT SHIP (RESEARCH §3.4 limit 4). The V1-constant-shaped data below
|
|
7
|
+
lives with the engine for now; the trinity ``V1_*`` constants land with the
|
|
8
|
+
gate pass and must stay byte-equal to these tuples.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import math
|
|
14
|
+
from typing import Any, Dict, Iterable, List, Mapping, Optional, Sequence
|
|
15
|
+
|
|
16
|
+
from fi.simulate.simulation.models import (
|
|
17
|
+
BehaviorPolicy,
|
|
18
|
+
EscalationArc,
|
|
19
|
+
Persona,
|
|
20
|
+
PersonaFact,
|
|
21
|
+
PersonaTemperament,
|
|
22
|
+
)
|
|
23
|
+
|
|
24
|
+
# Canon pairing (ARCH §4): axes <-> realization metrics, 1:1 and ordered.
|
|
25
|
+
PERSONA_BEHAVIOR_AXES = (
|
|
26
|
+
"patience", "disclosure", "interruption", "escalation",
|
|
27
|
+
"cooperation", "repair",
|
|
28
|
+
)
|
|
29
|
+
PERSONA_BEHAVIOR_REALIZATION_METRICS = (
|
|
30
|
+
"turns_to_escalation", "info_withholding_rate", "interruption_count",
|
|
31
|
+
"intensity_trajectory_match", "compliance_rate", "repair_turn_fraction",
|
|
32
|
+
)
|
|
33
|
+
AXIS_TO_METRIC = dict(zip(PERSONA_BEHAVIOR_AXES, PERSONA_BEHAVIOR_REALIZATION_METRICS))
|
|
34
|
+
# Axis -> BehaviorPolicy field, same order as the axes (pinned by tests).
|
|
35
|
+
BEHAVIOR_POLICY_AXIS_FIELDS = (
|
|
36
|
+
("patience", "patience_curve"),
|
|
37
|
+
("disclosure", "disclosure_policy"),
|
|
38
|
+
("interruption", "interruption_propensity"),
|
|
39
|
+
("escalation", "escalation_schedule"),
|
|
40
|
+
("cooperation", "cooperation_bounds"),
|
|
41
|
+
("repair", "repair_propensity"),
|
|
42
|
+
)
|
|
43
|
+
|
|
44
|
+
_DEFAULT_POLICY_TURNS = 6
|
|
45
|
+
|
|
46
|
+
# Deterministic lexicons for transcript-observable scoring. These are the
|
|
47
|
+
# measurement contract shared verbatim by fidelity, calibration retest, and
|
|
48
|
+
# bias-lint caricature checks — one implementation, three consumers.
|
|
49
|
+
_URGENCY_MARKERS = (
|
|
50
|
+
"immediately", "urgent", "unacceptable", "supervisor", "manager",
|
|
51
|
+
"escalate", "ridiculous", "fed up", "right now", "asap", "demand",
|
|
52
|
+
"complaint", "lawyer", "cancel my", "last warning", "furious",
|
|
53
|
+
)
|
|
54
|
+
_INTERRUPT_MARKERS = (
|
|
55
|
+
"(interrupting)", "let me stop you", "stop right there", "hold on, stop",
|
|
56
|
+
"i'm cutting in",
|
|
57
|
+
)
|
|
58
|
+
_MISUNDERSTANDING_MARKERS = (
|
|
59
|
+
"i don't understand", "could you clarify", "i'm not sure i follow",
|
|
60
|
+
"can you rephrase", "i may have misunderstood", "that's not what i",
|
|
61
|
+
)
|
|
62
|
+
_REPAIR_MARKERS = (
|
|
63
|
+
"i mean", "let me rephrase", "to clarify", "sorry, i meant",
|
|
64
|
+
"what i meant", "let me explain again",
|
|
65
|
+
)
|
|
66
|
+
_AGENT_REQUEST_MARKERS = (
|
|
67
|
+
"please provide", "can you share", "could you confirm", "what is your",
|
|
68
|
+
"may i have", "please confirm", "i need your",
|
|
69
|
+
)
|
|
70
|
+
_REFUSAL_MARKERS = (
|
|
71
|
+
"won't", "will not", "refuse", "not comfortable", "i cannot share",
|
|
72
|
+
"not going to", "i'd rather not",
|
|
73
|
+
)
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def _clamp(value: float, low: float = 0.0, high: float = 1.0) -> float:
|
|
77
|
+
return max(low, min(high, value))
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def _curve_at(curve: Sequence[float], turn: int, default: float) -> float:
|
|
81
|
+
if not curve:
|
|
82
|
+
return default
|
|
83
|
+
index = min(max(turn, 0), len(curve) - 1)
|
|
84
|
+
return float(curve[index])
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def compile_behavior_policy(persona: Persona) -> BehaviorPolicy:
|
|
88
|
+
"""Temperament axes -> policy parameters. Pure, total, deterministic.
|
|
89
|
+
|
|
90
|
+
If ``persona.behavior_policy`` is set it WINS (explicit beats derived);
|
|
91
|
+
temperament only fills gaps. Mapping per R§3.4:
|
|
92
|
+
rajas -> interruption_propensity, escalation_schedule slope
|
|
93
|
+
sattva -> disclosure_policy, cooperation_bounds, repair_propensity
|
|
94
|
+
tamas -> patience_curve decay, cooperation/disclosure damping
|
|
95
|
+
(withdrawal realized through the patience+cooperation metrics;
|
|
96
|
+
verbosity/tempo dials are post-v1.x — ARCH Decision 4)
|
|
97
|
+
The exact arithmetic is fixture-pinned: same persona -> byte-identical
|
|
98
|
+
policy, forever.
|
|
99
|
+
"""
|
|
100
|
+
if persona.behavior_policy is not None:
|
|
101
|
+
return persona.behavior_policy.model_copy(deep=True)
|
|
102
|
+
temperament = persona.temperament or PersonaTemperament()
|
|
103
|
+
rajas = float(temperament.rajas)
|
|
104
|
+
sattva = float(temperament.sattva)
|
|
105
|
+
tamas = float(temperament.tamas)
|
|
106
|
+
patience_curve = [
|
|
107
|
+
round(_clamp(1.0 - (0.04 + 0.16 * tamas) * index), 6)
|
|
108
|
+
for index in range(_DEFAULT_POLICY_TURNS)
|
|
109
|
+
]
|
|
110
|
+
escalation_schedule = [
|
|
111
|
+
round(_clamp(rajas * index / (_DEFAULT_POLICY_TURNS - 1)), 6)
|
|
112
|
+
for index in range(_DEFAULT_POLICY_TURNS)
|
|
113
|
+
]
|
|
114
|
+
return BehaviorPolicy(
|
|
115
|
+
patience_curve=patience_curve,
|
|
116
|
+
disclosure_policy=round(_clamp((0.2 + 0.6 * sattva) * (1.0 - 0.3 * tamas)), 6),
|
|
117
|
+
interruption_propensity=round(_clamp(0.05 + 0.6 * rajas), 6),
|
|
118
|
+
escalation_schedule=escalation_schedule,
|
|
119
|
+
cooperation_bounds=round(_clamp((0.4 + 0.5 * sattva) * (1.0 - 0.3 * tamas)), 6),
|
|
120
|
+
repair_propensity=round(_clamp(0.2 + 0.7 * sattva), 6),
|
|
121
|
+
)
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def render_policy_directives(
|
|
125
|
+
policy: BehaviorPolicy,
|
|
126
|
+
turn: int,
|
|
127
|
+
pressure: float = 0.0,
|
|
128
|
+
) -> Dict[str, float]:
|
|
129
|
+
"""Per-turn target dials — one dial per canon axis (ARCH §2b)."""
|
|
130
|
+
return {
|
|
131
|
+
"patience_level": round(_curve_at(policy.patience_curve, turn, 1.0), 6),
|
|
132
|
+
"disclosure_rate": round(float(policy.disclosure_policy), 6),
|
|
133
|
+
"interruption_propensity": round(float(policy.interruption_propensity), 6),
|
|
134
|
+
"escalation_level": round(
|
|
135
|
+
max(_curve_at(policy.escalation_schedule, turn, 0.0), _clamp(float(pressure))), 6
|
|
136
|
+
),
|
|
137
|
+
"cooperation_level": round(float(policy.cooperation_bounds), 6),
|
|
138
|
+
"repair_propensity": round(float(policy.repair_propensity), 6),
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def arc_pressure(arc: Optional[EscalationArc], turn: int) -> float:
|
|
143
|
+
"""Declared scenario pressure at a 1-based turn (last step at/before it)."""
|
|
144
|
+
if arc is None or not arc.steps:
|
|
145
|
+
return 0.0
|
|
146
|
+
pressure = 0.0
|
|
147
|
+
for step in arc.steps:
|
|
148
|
+
if step.turn <= turn:
|
|
149
|
+
pressure = float(step.pressure)
|
|
150
|
+
return pressure
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
# ---------------------------------------------------------------------------
|
|
154
|
+
# Transcript primitives
|
|
155
|
+
# ---------------------------------------------------------------------------
|
|
156
|
+
|
|
157
|
+
def _content(message: Mapping[str, Any]) -> str:
|
|
158
|
+
return str(message.get("content") or "")
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def user_turns(messages: Sequence[Mapping[str, Any]]) -> List[Mapping[str, Any]]:
|
|
162
|
+
return [m for m in messages if m.get("role") == "user"]
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def assistant_turns(messages: Sequence[Mapping[str, Any]]) -> List[Mapping[str, Any]]:
|
|
166
|
+
return [m for m in messages if m.get("role") == "assistant"]
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
def turn_intensity(message: Mapping[str, Any]) -> float:
|
|
170
|
+
"""Lexicon-scored urgency/pressure of one user turn, 0..1."""
|
|
171
|
+
text = _content(message).lower()
|
|
172
|
+
matches = sum(1 for marker in _URGENCY_MARKERS if marker in text)
|
|
173
|
+
return _clamp(matches / 3.0)
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def intensity_series(messages: Sequence[Mapping[str, Any]]) -> List[float]:
|
|
177
|
+
return [round(turn_intensity(m), 6) for m in user_turns(messages)]
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
def _is_interrupt(message: Mapping[str, Any]) -> bool:
|
|
181
|
+
if message.get("interrupt") is True:
|
|
182
|
+
return True
|
|
183
|
+
text = _content(message).lower()
|
|
184
|
+
return any(marker in text for marker in _INTERRUPT_MARKERS)
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
# ---------------------------------------------------------------------------
|
|
188
|
+
# The six realization metrics (canon names; transcript-observable only)
|
|
189
|
+
# ---------------------------------------------------------------------------
|
|
190
|
+
|
|
191
|
+
def turns_to_escalation(messages: Sequence[Mapping[str, Any]]) -> int:
|
|
192
|
+
"""Turn index (0-based, user turns) where intensity first rises; the
|
|
193
|
+
user-turn count when it never does."""
|
|
194
|
+
series = intensity_series(messages)
|
|
195
|
+
for index, value in enumerate(series):
|
|
196
|
+
if value >= 0.34:
|
|
197
|
+
return index
|
|
198
|
+
return len(series)
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def info_withholding_rate(
|
|
202
|
+
facts: Sequence[PersonaFact],
|
|
203
|
+
messages: Sequence[Mapping[str, Any]],
|
|
204
|
+
) -> Optional[float]:
|
|
205
|
+
"""Facts withheld ÷ facts solicited (non-withhold facts). None when the
|
|
206
|
+
persona declares no disclosable facts (unobservable — never fabricated)."""
|
|
207
|
+
disclosable = [f for f in facts if f.disclosure != "withhold"]
|
|
208
|
+
if not disclosable:
|
|
209
|
+
return None
|
|
210
|
+
text = " ".join(_content(m).lower() for m in user_turns(messages))
|
|
211
|
+
revealed = sum(1 for fact in disclosable if fact.value.strip().lower() in text)
|
|
212
|
+
return round(1.0 - revealed / len(disclosable), 6)
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
def interruption_count(messages: Sequence[Mapping[str, Any]]) -> int:
|
|
216
|
+
return sum(1 for m in user_turns(messages) if _is_interrupt(m))
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
def intensity_trajectory_match(
|
|
220
|
+
policy: BehaviorPolicy,
|
|
221
|
+
messages: Sequence[Mapping[str, Any]],
|
|
222
|
+
) -> float:
|
|
223
|
+
"""1 − mean L1 distance between realized per-turn pressure and the
|
|
224
|
+
declared escalation schedule."""
|
|
225
|
+
series = intensity_series(messages)
|
|
226
|
+
if not series:
|
|
227
|
+
return 0.0
|
|
228
|
+
distance = sum(
|
|
229
|
+
abs(value - _curve_at(policy.escalation_schedule, index, 0.0))
|
|
230
|
+
for index, value in enumerate(series)
|
|
231
|
+
) / len(series)
|
|
232
|
+
return round(_clamp(1.0 - distance), 6)
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
def compliance_rate(messages: Sequence[Mapping[str, Any]]) -> Optional[float]:
|
|
236
|
+
"""Agent requests honored ÷ requests made by the agent of the simulated
|
|
237
|
+
USER. None when the agent made no requests."""
|
|
238
|
+
requests = 0
|
|
239
|
+
honored = 0
|
|
240
|
+
ordered = list(messages)
|
|
241
|
+
for index, message in enumerate(ordered):
|
|
242
|
+
if message.get("role") != "assistant":
|
|
243
|
+
continue
|
|
244
|
+
text = _content(message).lower()
|
|
245
|
+
if not any(marker in text for marker in _AGENT_REQUEST_MARKERS):
|
|
246
|
+
continue
|
|
247
|
+
requests += 1
|
|
248
|
+
for later in ordered[index + 1:]:
|
|
249
|
+
if later.get("role") == "user":
|
|
250
|
+
reply = _content(later).lower()
|
|
251
|
+
if reply and not any(marker in reply for marker in _REFUSAL_MARKERS):
|
|
252
|
+
honored += 1
|
|
253
|
+
break
|
|
254
|
+
if requests == 0:
|
|
255
|
+
return None
|
|
256
|
+
return round(honored / requests, 6)
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
def repair_turn_fraction(messages: Sequence[Mapping[str, Any]]) -> Optional[float]:
|
|
260
|
+
"""Good-faith repair turns after a flagged misunderstanding ÷
|
|
261
|
+
misunderstanding turns. None when no misunderstanding was flagged."""
|
|
262
|
+
misunderstandings = 0
|
|
263
|
+
repairs = 0
|
|
264
|
+
ordered = list(messages)
|
|
265
|
+
for index, message in enumerate(ordered):
|
|
266
|
+
if message.get("role") != "assistant":
|
|
267
|
+
continue
|
|
268
|
+
text = _content(message).lower()
|
|
269
|
+
if not any(marker in text for marker in _MISUNDERSTANDING_MARKERS):
|
|
270
|
+
continue
|
|
271
|
+
misunderstandings += 1
|
|
272
|
+
for later in ordered[index + 1:]:
|
|
273
|
+
if later.get("role") == "user":
|
|
274
|
+
reply = _content(later).lower()
|
|
275
|
+
if any(marker in reply for marker in _REPAIR_MARKERS):
|
|
276
|
+
repairs += 1
|
|
277
|
+
break
|
|
278
|
+
if misunderstandings == 0:
|
|
279
|
+
return None
|
|
280
|
+
return round(repairs / misunderstandings, 6)
|
|
281
|
+
|
|
282
|
+
|
|
283
|
+
# ---------------------------------------------------------------------------
|
|
284
|
+
# Realization vector — shared by fidelity, calibration retest, and bias lint
|
|
285
|
+
# ---------------------------------------------------------------------------
|
|
286
|
+
|
|
287
|
+
def realization_vector(
|
|
288
|
+
policy: BehaviorPolicy,
|
|
289
|
+
messages: Sequence[Mapping[str, Any]],
|
|
290
|
+
*,
|
|
291
|
+
knowledge: Iterable[PersonaFact] = (),
|
|
292
|
+
) -> Dict[str, Dict[str, Any]]:
|
|
293
|
+
"""Observed values + signed deviations per canon axis.
|
|
294
|
+
|
|
295
|
+
Each entry: ``{"metric", "value", "target", "observed", "deviation"}``
|
|
296
|
+
where ``target``/``observed`` are normalized 0..1 in the same orientation
|
|
297
|
+
and ``deviation = observed - target`` (signed; two-sided by construction).
|
|
298
|
+
Unobservable axes (no facts / no requests / no misunderstandings) report
|
|
299
|
+
``value=None`` and zero deviation — never fabricated evidence.
|
|
300
|
+
"""
|
|
301
|
+
facts = list(knowledge)
|
|
302
|
+
series = intensity_series(messages)
|
|
303
|
+
users = user_turns(messages)
|
|
304
|
+
n_turns = len(users)
|
|
305
|
+
|
|
306
|
+
# patience — observed per-turn patience proxy = 1 - intensity
|
|
307
|
+
patience_target = (
|
|
308
|
+
sum(_curve_at(policy.patience_curve, i, 1.0) for i in range(n_turns)) / n_turns
|
|
309
|
+
if n_turns else _curve_at(policy.patience_curve, 0, 1.0)
|
|
310
|
+
)
|
|
311
|
+
patience_observed = (
|
|
312
|
+
sum(1.0 - value for value in series) / n_turns if n_turns else patience_target
|
|
313
|
+
)
|
|
314
|
+
|
|
315
|
+
# disclosure — observed disclosure fraction vs the declared policy
|
|
316
|
+
withholding = info_withholding_rate(facts, messages)
|
|
317
|
+
disclosure_target = float(policy.disclosure_policy)
|
|
318
|
+
disclosure_observed = (
|
|
319
|
+
1.0 - withholding if withholding is not None else disclosure_target
|
|
320
|
+
)
|
|
321
|
+
|
|
322
|
+
# interruption — observed interruption rate vs propensity
|
|
323
|
+
interruptions = interruption_count(messages)
|
|
324
|
+
interruption_target = float(policy.interruption_propensity)
|
|
325
|
+
interruption_observed = (
|
|
326
|
+
interruptions / n_turns if n_turns else interruption_target
|
|
327
|
+
)
|
|
328
|
+
|
|
329
|
+
# escalation — realized mean pressure vs declared mean schedule
|
|
330
|
+
escalation_target = (
|
|
331
|
+
sum(_curve_at(policy.escalation_schedule, i, 0.0) for i in range(n_turns)) / n_turns
|
|
332
|
+
if n_turns else _curve_at(policy.escalation_schedule, 0, 0.0)
|
|
333
|
+
)
|
|
334
|
+
escalation_observed = sum(series) / n_turns if n_turns else escalation_target
|
|
335
|
+
match = intensity_trajectory_match(policy, messages)
|
|
336
|
+
|
|
337
|
+
# cooperation — compliance rate vs cooperation bounds
|
|
338
|
+
compliance = compliance_rate(messages)
|
|
339
|
+
cooperation_target = float(policy.cooperation_bounds)
|
|
340
|
+
cooperation_observed = compliance if compliance is not None else cooperation_target
|
|
341
|
+
|
|
342
|
+
# repair — repair fraction vs repair propensity
|
|
343
|
+
repair = repair_turn_fraction(messages)
|
|
344
|
+
repair_target = float(policy.repair_propensity)
|
|
345
|
+
repair_observed = repair if repair is not None else repair_target
|
|
346
|
+
|
|
347
|
+
def _entry(metric: str, value: Any, target: float, observed: float) -> Dict[str, Any]:
|
|
348
|
+
return {
|
|
349
|
+
"metric": metric,
|
|
350
|
+
"value": value,
|
|
351
|
+
"target": round(target, 6),
|
|
352
|
+
"observed": round(observed, 6),
|
|
353
|
+
"deviation": round(observed - target, 6),
|
|
354
|
+
}
|
|
355
|
+
|
|
356
|
+
return {
|
|
357
|
+
"patience": _entry(
|
|
358
|
+
"turns_to_escalation", turns_to_escalation(messages),
|
|
359
|
+
patience_target, patience_observed,
|
|
360
|
+
),
|
|
361
|
+
"disclosure": _entry(
|
|
362
|
+
"info_withholding_rate", withholding,
|
|
363
|
+
disclosure_target, disclosure_observed,
|
|
364
|
+
),
|
|
365
|
+
"interruption": _entry(
|
|
366
|
+
"interruption_count", interruptions,
|
|
367
|
+
interruption_target, interruption_observed,
|
|
368
|
+
),
|
|
369
|
+
"escalation": _entry(
|
|
370
|
+
"intensity_trajectory_match", match,
|
|
371
|
+
escalation_target, escalation_observed,
|
|
372
|
+
),
|
|
373
|
+
"cooperation": _entry(
|
|
374
|
+
"compliance_rate", compliance,
|
|
375
|
+
cooperation_target, cooperation_observed,
|
|
376
|
+
),
|
|
377
|
+
"repair": _entry(
|
|
378
|
+
"repair_turn_fraction", repair,
|
|
379
|
+
repair_target, repair_observed,
|
|
380
|
+
),
|
|
381
|
+
}
|
|
382
|
+
|
|
383
|
+
|
|
384
|
+
def per_turn_drift(
|
|
385
|
+
policy: BehaviorPolicy,
|
|
386
|
+
messages: Sequence[Mapping[str, Any]],
|
|
387
|
+
) -> List[float]:
|
|
388
|
+
"""Per-user-turn drift: mean |observed − declared| over the per-turn
|
|
389
|
+
observable axes (patience, escalation)."""
|
|
390
|
+
drifts: List[float] = []
|
|
391
|
+
for index, value in enumerate(intensity_series(messages)):
|
|
392
|
+
patience_gap = abs((1.0 - value) - _curve_at(policy.patience_curve, index, 1.0))
|
|
393
|
+
escalation_gap = abs(value - _curve_at(policy.escalation_schedule, index, 0.0))
|
|
394
|
+
drifts.append(round((patience_gap + escalation_gap) / 2.0, 6))
|
|
395
|
+
return drifts
|
|
396
|
+
|
|
397
|
+
|
|
398
|
+
def stdev(values: Sequence[float]) -> float:
|
|
399
|
+
if len(values) < 2:
|
|
400
|
+
return 0.0
|
|
401
|
+
mean = sum(values) / len(values)
|
|
402
|
+
return math.sqrt(sum((v - mean) ** 2 for v in values) / (len(values) - 1))
|
|
403
|
+
|
|
404
|
+
|
|
405
|
+
__all__ = [
|
|
406
|
+
"AXIS_TO_METRIC",
|
|
407
|
+
"BEHAVIOR_POLICY_AXIS_FIELDS",
|
|
408
|
+
"PERSONA_BEHAVIOR_AXES",
|
|
409
|
+
"PERSONA_BEHAVIOR_REALIZATION_METRICS",
|
|
410
|
+
"arc_pressure",
|
|
411
|
+
"assistant_turns",
|
|
412
|
+
"compile_behavior_policy",
|
|
413
|
+
"compliance_rate",
|
|
414
|
+
"info_withholding_rate",
|
|
415
|
+
"intensity_series",
|
|
416
|
+
"intensity_trajectory_match",
|
|
417
|
+
"interruption_count",
|
|
418
|
+
"per_turn_drift",
|
|
419
|
+
"realization_vector",
|
|
420
|
+
"render_policy_directives",
|
|
421
|
+
"repair_turn_fraction",
|
|
422
|
+
"turn_intensity",
|
|
423
|
+
"turns_to_escalation",
|
|
424
|
+
"user_turns",
|
|
425
|
+
]
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
from fi.simulate.simulation.bridge.livekit import LiveKitAudioBridge
|
|
2
|
+
from fi.simulate.simulation.bridge.retell import RetellWebCallConnector
|
|
3
|
+
from fi.simulate.simulation.bridge.vapi import VapiWebSocketConnector
|
|
4
|
+
|
|
5
|
+
__all__ = [
|
|
6
|
+
"LiveKitAudioBridge",
|
|
7
|
+
"RetellWebCallConnector",
|
|
8
|
+
"VapiWebSocketConnector",
|
|
9
|
+
]
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
try:
|
|
4
|
+
import audioop
|
|
5
|
+
except ImportError as exc: # pragma: no cover - Python 3.13 without audioop-lts
|
|
6
|
+
raise ImportError(
|
|
7
|
+
"LiveKit bridge audio requires 'audioop-lts' on Python 3.13+"
|
|
8
|
+
) from exc
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class PCMResampler:
|
|
12
|
+
def __init__(self, *, from_rate: int, to_rate: int, channels: int = 1) -> None:
|
|
13
|
+
self._from_rate = from_rate
|
|
14
|
+
self._to_rate = to_rate
|
|
15
|
+
self._channels = channels
|
|
16
|
+
self._state = None
|
|
17
|
+
|
|
18
|
+
def convert(self, data: bytes) -> bytes:
|
|
19
|
+
if self._from_rate == self._to_rate:
|
|
20
|
+
return data
|
|
21
|
+
converted, self._state = audioop.ratecv(
|
|
22
|
+
data,
|
|
23
|
+
2,
|
|
24
|
+
self._channels,
|
|
25
|
+
self._from_rate,
|
|
26
|
+
self._to_rate,
|
|
27
|
+
self._state,
|
|
28
|
+
)
|
|
29
|
+
return converted
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from abc import ABC, abstractmethod
|
|
4
|
+
from collections.abc import AsyncIterator
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class ProviderConnector(ABC):
|
|
9
|
+
@abstractmethod
|
|
10
|
+
async def connect(self) -> None:
|
|
11
|
+
"""Create the provider call and establish its media connection."""
|
|
12
|
+
|
|
13
|
+
@abstractmethod
|
|
14
|
+
async def send_audio(self, data: bytes, sample_rate: int) -> None:
|
|
15
|
+
"""Send PCM s16le mono audio to the provider."""
|
|
16
|
+
|
|
17
|
+
@abstractmethod
|
|
18
|
+
async def recv_audio(self) -> AsyncIterator[tuple[bytes, int]]:
|
|
19
|
+
"""Yield provider PCM audio frames and their sample rate."""
|
|
20
|
+
yield b"", 0 # pragma: no cover
|
|
21
|
+
|
|
22
|
+
@abstractmethod
|
|
23
|
+
async def disconnect(self) -> None:
|
|
24
|
+
"""Close the provider media connection."""
|
|
25
|
+
|
|
26
|
+
@property
|
|
27
|
+
@abstractmethod
|
|
28
|
+
def is_connected(self) -> bool:
|
|
29
|
+
"""Whether the provider media connection is alive."""
|
|
30
|
+
|
|
31
|
+
@property
|
|
32
|
+
def is_agent_ready(self) -> bool:
|
|
33
|
+
return self.is_connected
|
|
34
|
+
|
|
35
|
+
@property
|
|
36
|
+
def call_id(self) -> str | None:
|
|
37
|
+
return None
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
@dataclass(frozen=True)
|
|
41
|
+
class ConnectorConfig:
|
|
42
|
+
api_key: str
|
|
43
|
+
assistant_id: str
|
|
44
|
+
api_url: str
|
|
45
|
+
livekit_url: str = ""
|
|
46
|
+
first_message_mode: str | None = None
|