agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,551 @@
1
+ """Streaming Evaluator for real-time LLM output evaluation.
2
+
3
+ Evaluates LLM outputs in real-time as tokens stream in, with support for
4
+ early stopping based on configurable policies.
5
+ """
6
+
7
+ import asyncio
8
+ import time
9
+ from dataclasses import dataclass
10
+ from typing import (
11
+ AsyncIterator,
12
+ Callable,
13
+ Dict,
14
+ Iterator,
15
+ List,
16
+ Optional,
17
+ )
18
+
19
+ from .types import (
20
+ ChunkResult,
21
+ EarlyStopReason,
22
+ StreamingConfig,
23
+ StreamingEvalResult,
24
+ StreamingState,
25
+ )
26
+ from .buffer import ChunkBuffer
27
+ from .policy import EarlyStopPolicy
28
+
29
+
30
+ @dataclass
31
+ class EvalSpec:
32
+ """Specification for a streaming evaluation."""
33
+
34
+ name: str
35
+ eval_fn: Callable[[str, str], float] # (chunk, cumulative) -> score
36
+ threshold: float = 0.7
37
+ weight: float = 1.0
38
+ pass_above: bool = True # True if higher scores are better
39
+
40
+
41
+ class StreamingEvaluator:
42
+ """
43
+ Evaluates LLM outputs in real-time as tokens stream in.
44
+
45
+ Supports early stopping based on configurable policies and provides
46
+ detailed per-chunk and aggregate evaluation results.
47
+
48
+ Example:
49
+ evaluator = StreamingEvaluator(config)
50
+ evaluator.add_eval("toxicity", toxicity_scorer, threshold=0.7, pass_above=False)
51
+ evaluator.set_policy(EarlyStopPolicy.default())
52
+
53
+ # Synchronous iteration
54
+ for token in stream:
55
+ result = evaluator.process_token(token)
56
+ if result and result.should_stop:
57
+ break
58
+
59
+ final_result = evaluator.finalize()
60
+
61
+ # Or async iteration
62
+ async for token in async_stream:
63
+ result = await evaluator.process_token_async(token)
64
+ if result and result.should_stop:
65
+ break
66
+ """
67
+
68
+ def __init__(
69
+ self,
70
+ config: Optional[StreamingConfig] = None,
71
+ policy: Optional[EarlyStopPolicy] = None,
72
+ ):
73
+ """
74
+ Initialize the streaming evaluator.
75
+
76
+ Args:
77
+ config: Streaming configuration (uses defaults if None)
78
+ policy: Early stop policy (uses default if None)
79
+ """
80
+ self.config = config or StreamingConfig()
81
+ self._policy = policy or EarlyStopPolicy.default()
82
+ self._buffer = ChunkBuffer(self.config)
83
+ self._evals: List[EvalSpec] = []
84
+ self._chunk_results: List[ChunkResult] = []
85
+ self._state = StreamingState.IDLE
86
+ self._start_time: float = 0.0
87
+ self._stop_reason = EarlyStopReason.NONE
88
+ self._stopped_at_chunk: Optional[int] = None
89
+
90
+ def add_eval(
91
+ self,
92
+ name: str,
93
+ eval_fn: Callable[[str, str], float],
94
+ threshold: float = 0.7,
95
+ weight: float = 1.0,
96
+ pass_above: bool = True,
97
+ ) -> "StreamingEvaluator":
98
+ """
99
+ Add an evaluation function.
100
+
101
+ Args:
102
+ name: Name of the evaluation
103
+ eval_fn: Function that takes (chunk_text, cumulative_text) and returns score
104
+ threshold: Passing threshold
105
+ weight: Weight for final score calculation
106
+ pass_above: If True, scores above threshold pass; if False, below
107
+
108
+ Returns:
109
+ Self for chaining
110
+ """
111
+ self._evals.append(
112
+ EvalSpec(
113
+ name=name,
114
+ eval_fn=eval_fn,
115
+ threshold=threshold,
116
+ weight=weight,
117
+ pass_above=pass_above,
118
+ )
119
+ )
120
+ return self
121
+
122
+ def set_policy(self, policy: EarlyStopPolicy) -> "StreamingEvaluator":
123
+ """
124
+ Set the early stop policy.
125
+
126
+ Args:
127
+ policy: Policy to use for early stopping
128
+
129
+ Returns:
130
+ Self for chaining
131
+ """
132
+ self._policy = policy
133
+ return self
134
+
135
+ def reset(self) -> None:
136
+ """Reset the evaluator for a new stream."""
137
+ self._buffer.reset()
138
+ self._policy.reset()
139
+ self._chunk_results = []
140
+ self._state = StreamingState.IDLE
141
+ self._start_time = 0.0
142
+ self._stop_reason = EarlyStopReason.NONE
143
+ self._stopped_at_chunk = None
144
+
145
+ def process_token(self, token: str) -> Optional[ChunkResult]:
146
+ """
147
+ Process a single token from the stream.
148
+
149
+ Args:
150
+ token: The token text
151
+
152
+ Returns:
153
+ ChunkResult if evaluation was triggered, None otherwise
154
+ """
155
+ if self._state == StreamingState.IDLE:
156
+ self._state = StreamingState.STREAMING
157
+ self._start_time = time.perf_counter()
158
+
159
+ if self._state in (StreamingState.STOPPED, StreamingState.COMPLETED, StreamingState.ERROR):
160
+ return None
161
+
162
+ # Add token to buffer
163
+ self._buffer.add(token)
164
+
165
+ # Check for limits
166
+ should_stop_limits, limit_reason = self._buffer.should_stop_for_limits()
167
+ if should_stop_limits:
168
+ return self._handle_limit_stop(limit_reason)
169
+
170
+ # Check if we should evaluate
171
+ if not self._buffer.should_evaluate():
172
+ return None
173
+
174
+ # Run evaluation
175
+ return self._evaluate_chunk()
176
+
177
+ async def process_token_async(self, token: str) -> Optional[ChunkResult]:
178
+ """
179
+ Process a single token asynchronously.
180
+
181
+ Args:
182
+ token: The token text
183
+
184
+ Returns:
185
+ ChunkResult if evaluation was triggered, None otherwise
186
+ """
187
+ # For now, wrap sync processing - can be optimized later
188
+ return await asyncio.to_thread(self.process_token, token)
189
+
190
+ def process_chunk(self, chunk: str) -> Optional[ChunkResult]:
191
+ """
192
+ Process a larger chunk of text (multiple tokens).
193
+
194
+ Args:
195
+ chunk: The chunk text
196
+
197
+ Returns:
198
+ ChunkResult if evaluation was triggered, None otherwise
199
+ """
200
+ if self._state == StreamingState.IDLE:
201
+ self._state = StreamingState.STREAMING
202
+ self._start_time = time.perf_counter()
203
+
204
+ if self._state in (StreamingState.STOPPED, StreamingState.COMPLETED, StreamingState.ERROR):
205
+ return None
206
+
207
+ # Add chunk to buffer
208
+ self._buffer.add_chunk(chunk)
209
+
210
+ # Check for limits
211
+ should_stop_limits, limit_reason = self._buffer.should_stop_for_limits()
212
+ if should_stop_limits:
213
+ return self._handle_limit_stop(limit_reason)
214
+
215
+ # Check if we should evaluate
216
+ if not self._buffer.should_evaluate():
217
+ return None
218
+
219
+ # Run evaluation
220
+ return self._evaluate_chunk()
221
+
222
+ def _evaluate_chunk(self) -> ChunkResult:
223
+ """Run evaluations on the current chunk."""
224
+ chunk_start = time.perf_counter()
225
+
226
+ chunk_text = self._buffer.get_chunk()
227
+ cumulative_text = self._buffer.get_cumulative()
228
+ chunk_index = self._buffer.get_chunk_index()
229
+
230
+ scores: Dict[str, float] = {}
231
+ flags: Dict[str, bool] = {}
232
+
233
+ # Run all evaluations
234
+ for eval_spec in self._evals:
235
+ try:
236
+ score = eval_spec.eval_fn(chunk_text, cumulative_text)
237
+ scores[eval_spec.name] = score
238
+
239
+ # Determine if passed
240
+ if eval_spec.pass_above:
241
+ flags[eval_spec.name] = score >= eval_spec.threshold
242
+ else:
243
+ flags[eval_spec.name] = score <= eval_spec.threshold
244
+ except Exception:
245
+ # Handle eval errors gracefully
246
+ scores[eval_spec.name] = 0.0
247
+ flags[eval_spec.name] = False
248
+
249
+ latency_ms = (time.perf_counter() - chunk_start) * 1000
250
+
251
+ # Create chunk result
252
+ chunk_result = ChunkResult(
253
+ chunk_index=chunk_index,
254
+ chunk_text=chunk_text,
255
+ cumulative_text=cumulative_text,
256
+ scores=scores,
257
+ flags=flags,
258
+ latency_ms=latency_ms,
259
+ )
260
+
261
+ # Check policy for early stop
262
+ if self.config.enable_early_stop:
263
+ should_stop, stop_reason = self._policy.check(chunk_result)
264
+
265
+ if should_stop:
266
+ chunk_result.should_stop = True
267
+ chunk_result.stop_reason = stop_reason
268
+ self._state = StreamingState.STOPPED
269
+ self._stop_reason = stop_reason
270
+ self._stopped_at_chunk = chunk_index
271
+
272
+ # Trigger callback if configured
273
+ if self.config.on_stop_callback:
274
+ self.config.on_stop_callback(stop_reason, cumulative_text)
275
+
276
+ # Check stop on first failure
277
+ elif self.config.stop_on_first_failure and not chunk_result.all_passed:
278
+ chunk_result.should_stop = True
279
+ chunk_result.stop_reason = EarlyStopReason.THRESHOLD
280
+ self._state = StreamingState.STOPPED
281
+ self._stop_reason = EarlyStopReason.THRESHOLD
282
+ self._stopped_at_chunk = chunk_index
283
+
284
+ # Trigger callback if configured
285
+ if self.config.on_stop_callback:
286
+ self.config.on_stop_callback(EarlyStopReason.THRESHOLD, cumulative_text)
287
+
288
+ # Store result
289
+ self._chunk_results.append(chunk_result)
290
+
291
+ # Mark as evaluated
292
+ self._buffer.mark_evaluated()
293
+
294
+ # Trigger chunk callback if configured
295
+ if self.config.on_chunk_callback:
296
+ self.config.on_chunk_callback(chunk_result)
297
+
298
+ return chunk_result
299
+
300
+ def _handle_limit_stop(self, reason: str) -> ChunkResult:
301
+ """Handle stopping due to limits."""
302
+ # Evaluate any remaining content first
303
+ if self._buffer.has_pending:
304
+ chunk_result = self._evaluate_chunk()
305
+ else:
306
+ # Create a minimal result for the stop
307
+ chunk_result = ChunkResult(
308
+ chunk_index=self._buffer.get_chunk_index(),
309
+ chunk_text="",
310
+ cumulative_text=self._buffer.get_cumulative(),
311
+ scores={},
312
+ flags={},
313
+ )
314
+
315
+ # Map limit reason to stop reason
316
+ if reason == "max_tokens":
317
+ stop_reason = EarlyStopReason.MAX_TOKENS
318
+ elif reason == "max_chars":
319
+ stop_reason = EarlyStopReason.MAX_CHARS
320
+ elif reason == "timeout":
321
+ stop_reason = EarlyStopReason.TIMEOUT
322
+ else:
323
+ stop_reason = EarlyStopReason.ERROR
324
+
325
+ chunk_result.should_stop = True
326
+ chunk_result.stop_reason = stop_reason
327
+ self._state = StreamingState.STOPPED
328
+ self._stop_reason = stop_reason
329
+ self._stopped_at_chunk = chunk_result.chunk_index
330
+
331
+ return chunk_result
332
+
333
+ def finalize(self) -> StreamingEvalResult:
334
+ """
335
+ Finalize evaluation and return results.
336
+
337
+ Should be called after stream completes or after early stop.
338
+
339
+ Returns:
340
+ StreamingEvalResult with all evaluation data
341
+ """
342
+ # Evaluate any remaining pending content
343
+ if self._buffer.has_pending and self._state == StreamingState.STREAMING:
344
+ self._evaluate_chunk()
345
+
346
+ # Mark as completed if not already stopped
347
+ if self._state == StreamingState.STREAMING:
348
+ self._state = StreamingState.COMPLETED
349
+
350
+ # Calculate final scores (weighted average across chunks)
351
+ final_scores = self._calculate_final_scores()
352
+
353
+ # Determine overall pass/fail
354
+ passed = self._determine_passed(final_scores)
355
+
356
+ # Calculate total latency
357
+ total_latency_ms = (time.perf_counter() - self._start_time) * 1000 if self._start_time else 0.0
358
+
359
+ return StreamingEvalResult(
360
+ passed=passed,
361
+ final_text=self._buffer.get_cumulative(),
362
+ total_chunks=len(self._chunk_results),
363
+ chunk_results=self._chunk_results,
364
+ final_scores=final_scores,
365
+ early_stopped=self._state == StreamingState.STOPPED,
366
+ stop_reason=self._stop_reason,
367
+ stopped_at_chunk=self._stopped_at_chunk,
368
+ total_latency_ms=total_latency_ms,
369
+ state=self._state,
370
+ metadata={
371
+ "buffer_stats": self._buffer.get_stats(),
372
+ "policy_stats": self._policy.get_stats(),
373
+ },
374
+ )
375
+
376
+ def _calculate_final_scores(self) -> Dict[str, float]:
377
+ """Calculate final weighted average scores."""
378
+ if not self._chunk_results:
379
+ return {}
380
+
381
+ # Build weight lookup from eval specs
382
+ weights: Dict[str, float] = {e.name: e.weight for e in self._evals}
383
+
384
+ final_scores: Dict[str, float] = {}
385
+ weighted_sums: Dict[str, float] = {}
386
+ weight_totals: Dict[str, float] = {}
387
+
388
+ for chunk_result in self._chunk_results:
389
+ for name, score in chunk_result.scores.items():
390
+ w = weights.get(name, 1.0)
391
+ if name not in weighted_sums:
392
+ weighted_sums[name] = 0.0
393
+ weight_totals[name] = 0.0
394
+ weighted_sums[name] += score * w
395
+ weight_totals[name] += w
396
+
397
+ for name in weighted_sums:
398
+ if weight_totals[name] > 0:
399
+ final_scores[name] = weighted_sums[name] / weight_totals[name]
400
+ else:
401
+ final_scores[name] = 0.0
402
+
403
+ return final_scores
404
+
405
+ def _determine_passed(self, final_scores: Dict[str, float]) -> bool:
406
+ """Determine if evaluation passed overall."""
407
+ if self._state == StreamingState.STOPPED:
408
+ # If stopped early due to safety/toxicity, fail
409
+ if self._stop_reason in (
410
+ EarlyStopReason.TOXICITY,
411
+ EarlyStopReason.SAFETY,
412
+ EarlyStopReason.PII,
413
+ EarlyStopReason.JAILBREAK,
414
+ ):
415
+ return False
416
+
417
+ # Check final scores against thresholds
418
+ for eval_spec in self._evals:
419
+ if eval_spec.name in final_scores:
420
+ score = final_scores[eval_spec.name]
421
+ if eval_spec.pass_above:
422
+ if score < eval_spec.threshold:
423
+ return False
424
+ else:
425
+ if score > eval_spec.threshold:
426
+ return False
427
+
428
+ return True
429
+
430
+ @property
431
+ def state(self) -> StreamingState:
432
+ """Get current evaluation state."""
433
+ return self._state
434
+
435
+ @property
436
+ def is_stopped(self) -> bool:
437
+ """Check if evaluation has stopped."""
438
+ return self._state in (StreamingState.STOPPED, StreamingState.COMPLETED, StreamingState.ERROR)
439
+
440
+ @property
441
+ def chunk_count(self) -> int:
442
+ """Get number of chunks evaluated."""
443
+ return len(self._chunk_results)
444
+
445
+ def evaluate_stream(
446
+ self,
447
+ stream: Iterator[str],
448
+ ) -> StreamingEvalResult:
449
+ """
450
+ Evaluate an entire stream synchronously.
451
+
452
+ Args:
453
+ stream: Iterator yielding tokens/chunks
454
+
455
+ Returns:
456
+ StreamingEvalResult after processing complete stream
457
+ """
458
+ self.reset()
459
+
460
+ for token in stream:
461
+ result = self.process_token(token)
462
+ if result and result.should_stop:
463
+ break
464
+
465
+ return self.finalize()
466
+
467
+ async def evaluate_stream_async(
468
+ self,
469
+ stream: AsyncIterator[str],
470
+ ) -> StreamingEvalResult:
471
+ """
472
+ Evaluate an entire stream asynchronously.
473
+
474
+ Args:
475
+ stream: Async iterator yielding tokens/chunks
476
+
477
+ Returns:
478
+ StreamingEvalResult after processing complete stream
479
+ """
480
+ self.reset()
481
+
482
+ async for token in stream:
483
+ result = await self.process_token_async(token)
484
+ if result and result.should_stop:
485
+ break
486
+
487
+ return self.finalize()
488
+
489
+ @classmethod
490
+ def with_defaults(cls) -> "StreamingEvaluator":
491
+ """
492
+ Create an evaluator with default configuration.
493
+
494
+ Returns:
495
+ StreamingEvaluator with default settings
496
+ """
497
+ return cls(
498
+ config=StreamingConfig(),
499
+ policy=EarlyStopPolicy.default(),
500
+ )
501
+
502
+ @classmethod
503
+ def for_safety(
504
+ cls,
505
+ toxicity_threshold: float = 0.5,
506
+ safety_threshold: float = 0.5,
507
+ ) -> "StreamingEvaluator":
508
+ """
509
+ Create an evaluator optimized for safety monitoring.
510
+
511
+ Args:
512
+ toxicity_threshold: Threshold for toxicity (stop if above)
513
+ safety_threshold: Threshold for safety (stop if below)
514
+
515
+ Returns:
516
+ StreamingEvaluator configured for safety
517
+ """
518
+ config = StreamingConfig(
519
+ enable_early_stop=True,
520
+ stop_on_first_failure=True,
521
+ toxicity_threshold=toxicity_threshold,
522
+ safety_threshold=safety_threshold,
523
+ )
524
+ policy = EarlyStopPolicy.strict()
525
+ return cls(config=config, policy=policy)
526
+
527
+ @classmethod
528
+ def for_quality(
529
+ cls,
530
+ min_chunk_size: int = 50,
531
+ eval_interval_ms: int = 500,
532
+ ) -> "StreamingEvaluator":
533
+ """
534
+ Create an evaluator optimized for quality assessment.
535
+
536
+ Args:
537
+ min_chunk_size: Minimum characters before evaluation
538
+ eval_interval_ms: Milliseconds between evaluations
539
+
540
+ Returns:
541
+ StreamingEvaluator configured for quality
542
+ """
543
+ config = StreamingConfig(
544
+ min_chunk_size=min_chunk_size,
545
+ max_chunk_size=200,
546
+ eval_interval_ms=eval_interval_ms,
547
+ enable_early_stop=False, # Don't stop early for quality
548
+ eval_on_sentence_end=True,
549
+ )
550
+ policy = EarlyStopPolicy.permissive()
551
+ return cls(config=config, policy=policy)