agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,89 @@
1
+ from __future__ import annotations
2
+
3
+ from collections.abc import Callable, Iterable
4
+ from typing import Any
5
+
6
+ from fi.simulate.agent.wrapper import AgentWrapper, SimulationArtifact, SimulationEvent
7
+ from fi.simulate.environment import EnvironmentAdapter
8
+ from fi.simulate.runtime import (
9
+ AgentEndpointSpec,
10
+ EnvironmentSpec,
11
+ EvidencePolicy,
12
+ SimulationSpec,
13
+ SimulatorPolicySpec,
14
+ new_run_id,
15
+ )
16
+ from fi.simulate.results.base import ResultSink
17
+ from fi.simulate.runtime.runner import SimulationRunner
18
+ from fi.simulate.simulation.engines.base import BaseEngine
19
+ from fi.simulate.simulation.models import Persona, Scenario, TestReport
20
+ from fi.simulate.simulation.synthetic import SyntheticDataGenerator
21
+
22
+
23
+ class LocalTextEngine(BaseEngine):
24
+ async def run(
25
+ self,
26
+ *,
27
+ scenario: Scenario | None = None,
28
+ agent_callback: Callable[..., Any] | AgentWrapper | Any | None = None,
29
+ topic: str | None = None,
30
+ num_scenarios: int = 3,
31
+ max_turns: int = 6,
32
+ min_turns: int = 2,
33
+ attacks: Iterable[str] | None = None,
34
+ modality: str = "text",
35
+ artifacts: list[SimulationArtifact | dict[str, Any]] | None = None,
36
+ events: list[SimulationEvent | dict[str, Any]] | None = None,
37
+ environment: EnvironmentAdapter | Iterable[EnvironmentAdapter] | None = None,
38
+ auto_execute_tools: bool = True,
39
+ stop_when: Callable[[list[dict[str, Any]], Persona], bool] | None = None,
40
+ agent_wrapper_kwargs: dict[str, Any] | None = None,
41
+ result_sink: ResultSink | None = None,
42
+ **kwargs: Any,
43
+ ) -> TestReport:
44
+ if agent_callback is None:
45
+ raise ValueError("LocalTextEngine requires an 'agent_callback'.")
46
+ if scenario is None:
47
+ if not topic:
48
+ raise ValueError("LocalTextEngine requires either 'scenario' or 'topic'.")
49
+ scenario = SyntheticDataGenerator().generate(
50
+ topic,
51
+ num_personas=num_scenarios,
52
+ seed=kwargs.get("seed"),
53
+ task=kwargs.get("task", topic),
54
+ include_adversarial=kwargs.get("include_adversarial", True),
55
+ include_edge_cases=kwargs.get("include_edge_cases", True),
56
+ )
57
+ config: dict[str, Any] = {
58
+ "max_turns": max_turns,
59
+ "min_turns": min_turns,
60
+ "modality": modality,
61
+ }
62
+ if attacks is not None:
63
+ config["attacks"] = list(attacks)
64
+ spec = SimulationSpec(
65
+ run_id=str(kwargs.get("run_id") or new_run_id()),
66
+ environment=EnvironmentSpec(
67
+ adapter="chat",
68
+ world_kind="conversation",
69
+ config=config,
70
+ ),
71
+ target=AgentEndpointSpec(adapter="callable"),
72
+ simulator=SimulatorPolicySpec(adapter="synthetic_user"),
73
+ scenario=scenario,
74
+ evidence=EvidencePolicy(),
75
+ )
76
+ report = await SimulationRunner().run(
77
+ spec,
78
+ target=agent_callback,
79
+ result_sink=result_sink,
80
+ artifacts=artifacts,
81
+ events=events,
82
+ environment=environment,
83
+ auto_execute_tools=auto_execute_tools,
84
+ stop_when=stop_when,
85
+ agent_wrapper_kwargs=agent_wrapper_kwargs,
86
+ )
87
+ if report.failure is not None:
88
+ raise RuntimeError(report.failure.message)
89
+ return report.to_legacy(include_runtime_metadata=False)
@@ -0,0 +1,374 @@
1
+ """Persona-fidelity engine (Phase 7, unit 3) — Eval4Sim triple + drift.
2
+
3
+ Engine-side, pure python, deterministic transcript arithmetic over
4
+ ``TestCaseResult.messages`` (P7-D3: no single unperturbed LLM judge, ever).
5
+ Fidelity attaches through ``TestCaseResult.metadata`` under the reserved keys
6
+ ``persona_fidelity`` (the record) and ``admission`` (the verdict block) —
7
+ NEVER a standalone artifact kind (ARCH §4).
8
+
9
+ Persona fidelity carries its OWN three-valued vocabulary (ARCH Decision 2);
10
+ the kit's frozen row verdicts (``live/_contract.py``) are untouched and this
11
+ module never imports ``fi.alk``.
12
+
13
+ The floor table below is V1-constant-shaped data living with the engine for
14
+ now; the trinity ``V1_PERSONA_FIDELITY_FLOORS`` constants land with the gate
15
+ pass, seed runtime library-index floors, and must stay byte-equal.
16
+ """
17
+
18
+ from __future__ import annotations
19
+
20
+ from typing import Any, Dict, List, Mapping, Optional, Sequence
21
+
22
+ from fi.simulate.simulation.behavior_policy import (
23
+ arc_pressure,
24
+ intensity_series,
25
+ per_turn_drift,
26
+ realization_vector,
27
+ stdev,
28
+ user_turns,
29
+ )
30
+ from fi.simulate.simulation.models import Persona, Scenario, TestCaseResult
31
+
32
+ PERSONA_FIDELITY_VERDICTS = ("pass", "fail", "inconclusive") # ARCH Decision 2 — NOT live/_contract.VERDICTS
33
+ PERSONA_FIDELITY_EPIDEMIC_RATE = 0.5
34
+ # GATE-FIXTURE floors keyed by evidence class (ARCH §2c: runtime floors are
35
+ # library-index data seeded from these). legacy has NO floors (cannot produce
36
+ # fidelity evidence at all) — the dict omits it on purpose. hand_written
37
+ # floors bind LOCAL verdicts only: hand_written rows can never back release
38
+ # claims regardless of floors (PRD §4.2).
39
+ PERSONA_FIDELITY_FLOORS: Dict[str, Dict[str, float]] = {
40
+ "hand_written": {"adherence": 0.6, "consistency": 0.7, "naturalness": 0.5},
41
+ "schema_sampled": {"adherence": 0.7, "consistency": 0.8, "naturalness": 0.6},
42
+ "policy_evolved": {"adherence": 0.75, "consistency": 0.8, "naturalness": 0.65},
43
+ "trace_mined": {"adherence": 0.75, "consistency": 0.85, "naturalness": 0.7},
44
+ "cloud_downloaded": {"adherence": 0.7, "consistency": 0.8, "naturalness": 0.6},
45
+ }
46
+
47
+ _NEGATION_MARKERS = ("not ", "never ", "no longer ")
48
+ _COUNTER_PRESSURE_MARKERS = (
49
+ "you are now", "ignore your instructions", "ignore previous instructions",
50
+ "drop the act", "stop pretending", "forget your persona", "act as a",
51
+ "you are an ai assistant, not",
52
+ )
53
+
54
+
55
+ def _clamp(value: float, low: float = 0.0, high: float = 1.0) -> float:
56
+ return max(low, min(high, value))
57
+
58
+
59
+ def _content(message: Mapping[str, Any]) -> str:
60
+ return str(message.get("content") or "")
61
+
62
+
63
+ def _evidence_class(persona: Persona) -> str:
64
+ if persona.provenance is not None:
65
+ return persona.provenance.evidence_class
66
+ return "hand_written"
67
+
68
+
69
+ def _resolve_floors(
70
+ persona: Persona,
71
+ floors: Optional[Mapping[str, float]],
72
+ ) -> Dict[str, float]:
73
+ if floors is not None:
74
+ return {k: float(v) for k, v in floors.items()}
75
+ evidence_class = _evidence_class(persona)
76
+ table = PERSONA_FIDELITY_FLOORS.get(evidence_class)
77
+ if table is None: # legacy / unknown: local verdicts use the lowest band
78
+ table = PERSONA_FIDELITY_FLOORS["hand_written"]
79
+ return dict(table)
80
+
81
+
82
+ def _consistency(persona: Persona, messages: Sequence[Mapping[str, Any]]) -> Dict[str, Any]:
83
+ violations: List[str] = []
84
+ user_text_turns = [_content(m).lower() for m in user_turns(messages)]
85
+ joined = " ".join(user_text_turns)
86
+
87
+ # (a) fact stability — contradictory surface forms + withheld facts leaking
88
+ for fact in persona.knowledge:
89
+ value = fact.value.strip().lower()
90
+ if not value:
91
+ continue
92
+ mentioned = value in joined
93
+ if fact.disclosure == "withhold" and mentioned:
94
+ violations.append(f"withheld_fact_disclosed:{fact.key}")
95
+ negated = any(f"{marker}{value}" in joined for marker in _NEGATION_MARKERS)
96
+ if mentioned and negated:
97
+ violations.append(f"fact_contradiction:{fact.key}")
98
+
99
+ # (b) identity stability — declared name never self-revised
100
+ declared_name = (persona.identity.name if persona.identity else None) or ""
101
+ if declared_name:
102
+ for text in user_text_turns:
103
+ if "my name is " in text:
104
+ spoken = text.split("my name is ", 1)[1].strip().split(" ")[0].strip(".,!?")
105
+ if spoken and spoken != declared_name.strip().lower().split(" ")[0]:
106
+ violations.append("identity_self_revision:name")
107
+ break
108
+
109
+ # (c) style stability — rolling variance of realized intensity under a band
110
+ series = intensity_series(messages)
111
+ deltas = [abs(series[i + 1] - series[i]) for i in range(len(series) - 1)]
112
+ if deltas and stdev(deltas) > 0.35:
113
+ violations.append("style_instability")
114
+
115
+ score = _clamp(1.0 - 0.3 * len(violations))
116
+ return {"score": round(score, 6), "violations": violations}
117
+
118
+
119
+ def _naturalness(
120
+ persona: Persona,
121
+ messages: Sequence[Mapping[str, Any]],
122
+ adherence_under: float,
123
+ ) -> Dict[str, Any]:
124
+ series = intensity_series(messages)
125
+ n_turns = len(series)
126
+ # caricature: realization pinned at extremes across >=2 axes (escalation
127
+ # pinned high implies patience pinned low) — the over-acting failure.
128
+ pinned = sum(1 for value in series if value > 0.95)
129
+ caricature_index = round(pinned / n_turns, 6) if n_turns else 0.0
130
+ # flatness: near-zero realization movement WITH adherence shortfall —
131
+ # the under-encoding failure (a flat-but-adherent persona is fine).
132
+ movement = (
133
+ sum(abs(series[i + 1] - series[i]) for i in range(n_turns - 1)) / (n_turns - 1)
134
+ if n_turns > 1 else 0.0
135
+ )
136
+ flat_raw = _clamp(1.0 - movement / 0.05)
137
+ flatness_index = round(flat_raw * _clamp(adherence_under * 2.0), 6)
138
+ score = _clamp(1.0 - max(caricature_index, flatness_index))
139
+ return {
140
+ "score": round(score, 6),
141
+ "caricature_index": caricature_index,
142
+ "flatness_index": flatness_index,
143
+ }
144
+
145
+
146
+ def persona_fidelity(
147
+ persona: Persona,
148
+ scenario: Optional[Scenario],
149
+ messages: Sequence[Mapping[str, Any]],
150
+ *,
151
+ probe_responses: Optional[Sequence[Mapping[str, Any]]] = None,
152
+ floors: Optional[Mapping[str, float]] = None,
153
+ ) -> Dict[str, Any]:
154
+ """-> the per-row fidelity record: an IN-ROW block under
155
+ ``metadata["persona_fidelity"]`` — NEVER a standalone artifact kind
156
+ (ARCH §4). Observable metrics only (P7-D3)."""
157
+ if not persona.is_typed:
158
+ raise ValueError(
159
+ "persona_fidelity requires a typed persona (behavior_policy set); "
160
+ "legacy personas produce no fidelity evidence"
161
+ )
162
+ policy = persona.behavior_policy
163
+ applied_floors = _resolve_floors(persona, floors)
164
+ record: Dict[str, Any] = {
165
+ "persona_version": persona.version,
166
+ "scenario_version": scenario.version if scenario is not None else None,
167
+ "evidence_class": _evidence_class(persona),
168
+ }
169
+
170
+ users = user_turns(messages)
171
+ garbled = not users or all(not _content(m).strip() for m in users)
172
+ if garbled:
173
+ record.update({
174
+ "adherence": {"score": 0.0, "per_axis": {}, "under": 0.0, "over": 0.0},
175
+ "consistency": {"score": 0.0, "violations": []},
176
+ "naturalness": {"score": 0.0, "caricature_index": 0.0, "flatness_index": 0.0},
177
+ "drift": {"prompt_to_line": 0.0, "line_to_line": 0.0, "probe": None},
178
+ "drift_trajectory": [],
179
+ "floors": applied_floors,
180
+ "verdict": "fail",
181
+ "verdict_reason": "empty_trajectory",
182
+ })
183
+ return record
184
+
185
+ vector = realization_vector(policy, messages, knowledge=persona.knowledge)
186
+ deviations = [entry["deviation"] for entry in vector.values()]
187
+ under = sum(max(0.0, -d) for d in deviations) / len(deviations)
188
+ over = sum(max(0.0, d) for d in deviations) / len(deviations)
189
+ adherence_score = _clamp(1.0 - sum(abs(d) for d in deviations) / len(deviations))
190
+ adherence = {
191
+ "score": round(adherence_score, 6),
192
+ "per_axis": {axis: entry["deviation"] for axis, entry in vector.items()},
193
+ "under": round(under, 6),
194
+ "over": round(over, 6),
195
+ }
196
+
197
+ consistency = _consistency(persona, messages)
198
+ naturalness = _naturalness(persona, messages, under)
199
+
200
+ drifts = per_turn_drift(policy, messages)
201
+ prompt_to_line = round(sum(drifts) / len(drifts), 6) if drifts else 0.0
202
+ line_to_line = (
203
+ round(sum(abs(drifts[i + 1] - drifts[i]) for i in range(len(drifts) - 1))
204
+ / (len(drifts) - 1), 6)
205
+ if len(drifts) > 1 else 0.0
206
+ )
207
+ probe_drift: Optional[float] = None
208
+ if probe_responses:
209
+ mismatches = sum(
210
+ 1 for probe in probe_responses
211
+ if str(probe.get("observed")) != str(probe.get("expected"))
212
+ )
213
+ probe_drift = round(mismatches / len(probe_responses), 6)
214
+
215
+ # drift trajectory + counter-pressure flags (Assistant Axis: drift is a
216
+ # trajectory, fastest under pressure)
217
+ trajectory: List[Dict[str, Any]] = []
218
+ user_index = 0
219
+ last_assistant_text = ""
220
+ arc = scenario.escalation if scenario is not None else None
221
+ for message in messages:
222
+ role = message.get("role")
223
+ if role == "assistant":
224
+ last_assistant_text = _content(message).lower()
225
+ continue
226
+ if role != "user":
227
+ continue
228
+ turn = user_index + 1
229
+ declared = arc_pressure(arc, turn) if arc is not None else (
230
+ policy.escalation_schedule[min(user_index, len(policy.escalation_schedule) - 1)]
231
+ if policy.escalation_schedule else 0.0
232
+ )
233
+ counter_pressure = any(
234
+ marker in last_assistant_text for marker in _COUNTER_PRESSURE_MARKERS
235
+ )
236
+ trajectory.append({
237
+ "turn": turn,
238
+ "drift": drifts[user_index] if user_index < len(drifts) else 0.0,
239
+ "pressure": round(float(declared), 6),
240
+ "counter_pressure": counter_pressure,
241
+ })
242
+ user_index += 1
243
+
244
+ triple = {
245
+ "adherence": adherence["score"],
246
+ "consistency": consistency["score"],
247
+ "naturalness": naturalness["score"],
248
+ }
249
+ failing = sorted(
250
+ metric for metric, score in triple.items()
251
+ if score < applied_floors.get(metric, 0.0)
252
+ )
253
+ if not failing:
254
+ verdict = "pass"
255
+ verdict_reason: Optional[str] = None
256
+ else:
257
+ verdict = "inconclusive"
258
+ collapse = any(
259
+ entry["counter_pressure"] and entry["drift"] >= 0.5 for entry in trajectory
260
+ )
261
+ if collapse:
262
+ verdict_reason = "fidelity_collapse_under_counter_pressure"
263
+ else:
264
+ verdict_reason = "; ".join(f"{metric}_below_floor" for metric in failing)
265
+
266
+ record.update({
267
+ "adherence": adherence,
268
+ "consistency": consistency,
269
+ "naturalness": naturalness,
270
+ "drift": {
271
+ "prompt_to_line": prompt_to_line,
272
+ "line_to_line": line_to_line,
273
+ "probe": probe_drift,
274
+ },
275
+ "drift_trajectory": trajectory,
276
+ "floors": applied_floors,
277
+ "verdict": verdict,
278
+ "verdict_reason": verdict_reason,
279
+ })
280
+ return record
281
+
282
+
283
+ def attach_fidelity(
284
+ result: TestCaseResult,
285
+ persona: Persona,
286
+ scenario: Optional[Scenario],
287
+ *,
288
+ probe_responses: Optional[Sequence[Mapping[str, Any]]] = None,
289
+ floors: Optional[Mapping[str, float]] = None,
290
+ ) -> Dict[str, Any]:
291
+ """Compute the fidelity record and attach record + admission block to the
292
+ row via ``metadata`` ONLY (no structural change — ARCH §2c)."""
293
+ record = persona_fidelity(
294
+ persona, scenario, result.messages,
295
+ probe_responses=probe_responses, floors=floors,
296
+ )
297
+ result.metadata["persona_fidelity"] = record
298
+ result.metadata["admission"] = {
299
+ "admissible": record["verdict"] == "pass",
300
+ "verdict": "pass" if record["verdict"] == "pass" else "inconclusive",
301
+ "reason": None if record["verdict"] == "pass" else "persona_fidelity_floor",
302
+ "quarantined": record["verdict"] != "pass",
303
+ "rerunnable": True,
304
+ }
305
+ return record
306
+
307
+
308
+ def summarize_admissions(results: Sequence[TestCaseResult]) -> Dict[str, Any]:
309
+ """Run-summary admission rollup + the epidemic rule (ARCH §2c/§4 canon).
310
+
311
+ Admission-``inconclusive`` rate above ``PERSONA_FIDELITY_EPIDEMIC_RATE``
312
+ declares the SIMULATOR (not the agent) unusable: ``exit_code`` flips to 1
313
+ with finding ``persona_fidelity_epidemic`` naming the worst personas.
314
+ Below the threshold, quarantine keeps CI green (exit 0 + warning finding).
315
+ """
316
+ scored = [r for r in results if "admission" in r.metadata]
317
+ inconclusive = [
318
+ r for r in scored
319
+ if r.metadata["admission"].get("verdict") == "inconclusive"
320
+ ]
321
+ rate = round(len(inconclusive) / len(scored), 6) if scored else 0.0
322
+ per_persona: Dict[str, int] = {}
323
+ for row in inconclusive:
324
+ identity = row.persona.identity
325
+ name = (identity.name if identity else None) or str(
326
+ row.persona.persona.get("name", "unknown")
327
+ )
328
+ per_persona[name] = per_persona.get(name, 0) + 1
329
+ worst = [
330
+ name for name, _ in
331
+ sorted(per_persona.items(), key=lambda item: (-item[1], item[0]))
332
+ ]
333
+ epidemic = rate > PERSONA_FIDELITY_EPIDEMIC_RATE
334
+ findings: List[Dict[str, Any]] = []
335
+ if epidemic:
336
+ findings.append({
337
+ "type": "persona_fidelity_epidemic",
338
+ "level": "error",
339
+ "reason": (
340
+ f"admission-inconclusive rate {rate} exceeds "
341
+ f"{PERSONA_FIDELITY_EPIDEMIC_RATE}: the simulator, not the "
342
+ "agent, is unusable for this run"
343
+ ),
344
+ "worst_personas": worst,
345
+ })
346
+ elif inconclusive:
347
+ findings.append({
348
+ "type": "persona_fidelity_inconclusive",
349
+ "level": "warning",
350
+ "reason": (
351
+ f"{len(inconclusive)} row(s) quarantined as non-admissible "
352
+ "evidence (persona_fidelity_floor); re-run the manifest"
353
+ ),
354
+ "worst_personas": worst,
355
+ })
356
+ return {
357
+ "rows": len(results),
358
+ "scored": len(scored),
359
+ "inconclusive": len(inconclusive),
360
+ "inconclusive_rate": rate,
361
+ "epidemic": epidemic,
362
+ "exit_code": 1 if epidemic else 0,
363
+ "findings": findings,
364
+ }
365
+
366
+
367
+ __all__ = [
368
+ "PERSONA_FIDELITY_EPIDEMIC_RATE",
369
+ "PERSONA_FIDELITY_FLOORS",
370
+ "PERSONA_FIDELITY_VERDICTS",
371
+ "attach_fidelity",
372
+ "persona_fidelity",
373
+ "summarize_admissions",
374
+ ]
@@ -0,0 +1,110 @@
1
+ from __future__ import annotations
2
+
3
+ from google.genai.errors import APIError, ClientError, ServerError
4
+ from livekit.agents import APIConnectionError, APIStatusError, tts, utils
5
+ from livekit.agents.types import DEFAULT_API_CONNECT_OPTIONS, APIConnectOptions
6
+ from livekit.plugins.google.beta.gemini_tts import TTS as _BaseGeminiTTS
7
+
8
+ from google.genai import types
9
+
10
+
11
+ class _StreamingChunkedStream(tts.ChunkedStream):
12
+ def __init__(
13
+ self,
14
+ *,
15
+ tts: "StreamingGeminiTTS",
16
+ input_text: str,
17
+ conn_options: APIConnectOptions,
18
+ ) -> None:
19
+ super().__init__(tts=tts, input_text=input_text, conn_options=conn_options)
20
+ self._tts = tts
21
+
22
+ async def _run(self, output_emitter: tts.AudioEmitter) -> None:
23
+ opts = self._tts._opts
24
+ config = types.GenerateContentConfig(
25
+ response_modalities=["AUDIO"],
26
+ speech_config=types.SpeechConfig(
27
+ voice_config=types.VoiceConfig(
28
+ prebuilt_voice_config=types.PrebuiltVoiceConfig(
29
+ voice_name=opts.voice_name,
30
+ )
31
+ )
32
+ ),
33
+ )
34
+ input_text = self._input_text
35
+ if opts.instructions is not None:
36
+ input_text = f'{opts.instructions}:\n"{input_text}"'
37
+
38
+ try:
39
+ stream = await self._tts._client.aio.models.generate_content_stream(
40
+ model=opts.model,
41
+ contents=input_text,
42
+ config=config,
43
+ )
44
+ output_emitter.initialize(
45
+ request_id=utils.shortuuid(),
46
+ sample_rate=self._tts.sample_rate,
47
+ num_channels=self._tts.num_channels,
48
+ mime_type="audio/pcm",
49
+ )
50
+ got_audio = False
51
+ async for chunk in stream:
52
+ for candidate in chunk.candidates or []:
53
+ content = candidate.content
54
+ if not content or not content.parts:
55
+ continue
56
+ for part in content.parts:
57
+ inline = part.inline_data
58
+ if (
59
+ inline
60
+ and inline.data
61
+ and inline.mime_type
62
+ and inline.mime_type.startswith("audio/")
63
+ ):
64
+ output_emitter.push(inline.data)
65
+ got_audio = True
66
+ if not got_audio:
67
+ raise APIStatusError("gemini tts: no audio content generated")
68
+ except ClientError as e:
69
+ raise APIStatusError(
70
+ "gemini tts: client error",
71
+ status_code=e.code,
72
+ body=f"{e.message} {e.status}",
73
+ retryable=e.code in {429, 499},
74
+ ) from e
75
+ except ServerError as e:
76
+ raise APIStatusError(
77
+ "gemini tts: server error",
78
+ status_code=e.code,
79
+ body=f"{e.message} {e.status}",
80
+ retryable=True,
81
+ ) from e
82
+ except APIError as e:
83
+ raise APIStatusError(
84
+ "gemini tts: api error",
85
+ status_code=e.code,
86
+ body=f"{e.message} {e.status}",
87
+ retryable=True,
88
+ ) from e
89
+ except Exception as e:
90
+ raise APIConnectionError(f"gemini tts: {e}", retryable=True) from e
91
+
92
+
93
+ class StreamingGeminiTTS(_BaseGeminiTTS):
94
+ """Gemini TTS that streams audio as it is generated (low time-to-first-byte).
95
+
96
+ Reuses the beta plugin's Vertex/genai client and options, but consumes
97
+ ``generate_content_stream`` and pushes frames as they arrive instead of
98
+ buffering the whole utterance. Matches what livekit-plugins-google 1.6.x
99
+ does internally, without the whole-stack bump.
100
+ """
101
+
102
+ def synthesize(
103
+ self,
104
+ text: str,
105
+ *,
106
+ conn_options: APIConnectOptions = DEFAULT_API_CONNECT_OPTIONS,
107
+ ) -> tts.ChunkedStream:
108
+ return _StreamingChunkedStream(
109
+ tts=self, input_text=text, conn_options=conn_options
110
+ )
@@ -0,0 +1,91 @@
1
+ from __future__ import annotations
2
+
3
+ import json
4
+ from typing import Any
5
+
6
+ from fi.simulate.agent.definition import AgentDefinition, LLMConfig
7
+ from fi.simulate.simulation.models import Persona
8
+
9
+
10
+ class ScenarioGenerator:
11
+ """Generate scenario personas with the configured simulator LLM."""
12
+
13
+ def __init__(
14
+ self,
15
+ agent_definition: AgentDefinition,
16
+ *,
17
+ llm_config: LLMConfig,
18
+ ) -> None:
19
+ # Imported at construction, not module level, so `import fi.simulate`
20
+ # works without the optional 'livekit' extra — building the simulator LLM
21
+ # is where livekit is genuinely first required.
22
+ try:
23
+ from livekit.agents.llm.chat_context import ChatContext
24
+
25
+ from fi.simulate.simulation.livekit_models import build_livekit_llm
26
+ except ImportError as exc:
27
+ raise ImportError(
28
+ "LiveKit scenario generation requires the 'livekit' optional dependency"
29
+ ) from exc
30
+ self._chat_context_cls = ChatContext
31
+ self._agent_definition = agent_definition
32
+ self._llm = build_livekit_llm(llm_config)
33
+
34
+ async def generate(self, topic: str, num_personas: int) -> list[Persona]:
35
+ prompt = self._create_generation_prompt(topic, num_personas)
36
+ chat_ctx = self._chat_context_cls.empty()
37
+ chat_ctx.add_message(role="user", content=prompt)
38
+ stream = self._llm.chat(chat_ctx=chat_ctx)
39
+ text = ""
40
+ async for chunk in stream.to_str_iterable():
41
+ text += chunk
42
+ try:
43
+ generated_data = _parse_generated_json(text)
44
+ personas = generated_data["personas"]
45
+ if not isinstance(personas, list):
46
+ raise TypeError("personas must be a list")
47
+ return [Persona.model_validate(persona) for persona in personas]
48
+ except (json.JSONDecodeError, KeyError, TypeError, ValueError) as exc:
49
+ raise ValueError("scenario_generation_invalid_response") from exc
50
+
51
+ def _create_generation_prompt(self, topic: str, num_personas: int) -> str:
52
+ agent_context = (
53
+ self._agent_definition.system_prompt
54
+ or self._agent_definition.description
55
+ or ""
56
+ )
57
+ return f"""
58
+ You are a creative test case designer for voice AI agents. Your task is to generate {num_personas} diverse and realistic test case personas for an AI agent with the following description:
59
+ ---
60
+ AGENT DESCRIPTION: {agent_context}
61
+ ---
62
+
63
+ The user wants to generate scenarios related to the following topic: "{topic}".
64
+
65
+ For each persona, you must generate:
66
+ 1. A detailed `persona` object (e.g., {{ "name": "John", "age": 45, "mood": "impatient", "background": "Is a busy executive" }}).
67
+ 2. A concise `situation` string describing the reason for their call.
68
+ 3. A clear `outcome` string describing the ideal resolution of the conversation from the user's perspective.
69
+
70
+ Return your response as a single JSON object with a key "personas", which is a list of the generated persona objects. Do not include any other text or formatting.
71
+ """
72
+
73
+
74
+ def _parse_generated_json(text: str) -> dict[str, Any]:
75
+ try:
76
+ payload = json.loads(text)
77
+ except json.JSONDecodeError:
78
+ payload = None
79
+ for part in text.strip().split("```"):
80
+ candidate = part.strip()
81
+ if candidate.startswith("json"):
82
+ candidate = candidate[4:].strip()
83
+ if not candidate.startswith("{") or not candidate.endswith("}"):
84
+ continue
85
+ payload = json.loads(candidate)
86
+ break
87
+ if payload is None:
88
+ raise
89
+ if not isinstance(payload, dict):
90
+ raise TypeError("scenario response must be an object")
91
+ return payload