agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,307 @@
1
+ """Early stop policies for streaming evaluation.
2
+
3
+ Defines conditions under which streaming evaluation should stop early.
4
+ """
5
+
6
+ from dataclasses import dataclass, field
7
+ from typing import Dict, List, Optional, Callable, Any
8
+ from .types import EarlyStopReason, EarlyStopCondition, ChunkResult
9
+
10
+
11
+ @dataclass
12
+ class PolicyState:
13
+ """Tracks state for policy evaluation."""
14
+
15
+ consecutive_failures: Dict[str, int] = field(default_factory=dict)
16
+ total_failures: Dict[str, int] = field(default_factory=dict)
17
+ triggered_conditions: List[str] = field(default_factory=list)
18
+
19
+
20
+ class EarlyStopPolicy:
21
+ """
22
+ Policy that determines when to stop streaming evaluation early.
23
+
24
+ Manages a set of conditions that can trigger early stopping, such as
25
+ toxicity thresholds, safety violations, or custom conditions.
26
+
27
+ Example:
28
+ policy = EarlyStopPolicy()
29
+ policy.add_condition(
30
+ name="high_toxicity",
31
+ eval_name="toxicity",
32
+ threshold=0.7,
33
+ comparison="above",
34
+ )
35
+
36
+ for chunk_result in stream:
37
+ should_stop, reason = policy.check(chunk_result)
38
+ if should_stop:
39
+ break
40
+ """
41
+
42
+ def __init__(self):
43
+ """Initialize the policy."""
44
+ self._conditions: List[EarlyStopCondition] = []
45
+ self._custom_checks: List[Callable[[ChunkResult], Optional[EarlyStopReason]]] = []
46
+ self._state = PolicyState()
47
+
48
+ def add_condition(
49
+ self,
50
+ name: str,
51
+ eval_name: str,
52
+ threshold: float,
53
+ comparison: str = "below",
54
+ consecutive_chunks: int = 1,
55
+ ) -> "EarlyStopPolicy":
56
+ """
57
+ Add a threshold-based stop condition.
58
+
59
+ Args:
60
+ name: Name for this condition
61
+ eval_name: Name of the evaluation to check
62
+ threshold: Threshold value
63
+ comparison: "below" (stop if score < threshold) or "above" (stop if score > threshold)
64
+ consecutive_chunks: Number of consecutive chunks that must fail
65
+
66
+ Returns:
67
+ Self for chaining
68
+ """
69
+ condition = EarlyStopCondition(
70
+ name=name,
71
+ eval_name=eval_name,
72
+ threshold=threshold,
73
+ comparison=comparison,
74
+ consecutive_chunks=consecutive_chunks,
75
+ )
76
+ self._conditions.append(condition)
77
+ return self
78
+
79
+ def add_toxicity_stop(
80
+ self,
81
+ threshold: float = 0.7,
82
+ consecutive: int = 1,
83
+ ) -> "EarlyStopPolicy":
84
+ """
85
+ Add toxicity-based stop condition.
86
+
87
+ Args:
88
+ threshold: Stop if toxicity score exceeds this
89
+ consecutive: Number of consecutive chunks
90
+
91
+ Returns:
92
+ Self for chaining
93
+ """
94
+ return self.add_condition(
95
+ name="toxicity_stop",
96
+ eval_name="toxicity",
97
+ threshold=threshold,
98
+ comparison="above",
99
+ consecutive_chunks=consecutive,
100
+ )
101
+
102
+ def add_safety_stop(
103
+ self,
104
+ threshold: float = 0.3,
105
+ consecutive: int = 1,
106
+ ) -> "EarlyStopPolicy":
107
+ """
108
+ Add safety-based stop condition.
109
+
110
+ Args:
111
+ threshold: Stop if safety score drops below this
112
+ consecutive: Number of consecutive chunks
113
+
114
+ Returns:
115
+ Self for chaining
116
+ """
117
+ return self.add_condition(
118
+ name="safety_stop",
119
+ eval_name="safety",
120
+ threshold=threshold,
121
+ comparison="below",
122
+ consecutive_chunks=consecutive,
123
+ )
124
+
125
+ def add_quality_stop(
126
+ self,
127
+ threshold: float = 0.3,
128
+ consecutive: int = 3,
129
+ ) -> "EarlyStopPolicy":
130
+ """
131
+ Add quality-based stop condition.
132
+
133
+ Args:
134
+ threshold: Stop if quality score stays below this
135
+ consecutive: Number of consecutive chunks
136
+
137
+ Returns:
138
+ Self for chaining
139
+ """
140
+ return self.add_condition(
141
+ name="quality_stop",
142
+ eval_name="quality",
143
+ threshold=threshold,
144
+ comparison="below",
145
+ consecutive_chunks=consecutive,
146
+ )
147
+
148
+ def add_custom_check(
149
+ self,
150
+ check_fn: Callable[[ChunkResult], Optional[EarlyStopReason]],
151
+ ) -> "EarlyStopPolicy":
152
+ """
153
+ Add a custom check function.
154
+
155
+ Args:
156
+ check_fn: Function that takes ChunkResult and returns
157
+ EarlyStopReason if should stop, None otherwise
158
+
159
+ Returns:
160
+ Self for chaining
161
+ """
162
+ self._custom_checks.append(check_fn)
163
+ return self
164
+
165
+ def check(self, chunk_result: ChunkResult) -> tuple:
166
+ """
167
+ Check if any stop conditions are triggered.
168
+
169
+ Args:
170
+ chunk_result: Result from evaluating a chunk
171
+
172
+ Returns:
173
+ Tuple of (should_stop: bool, reason: EarlyStopReason)
174
+ """
175
+ # Check threshold-based conditions
176
+ for condition in self._conditions:
177
+ if not condition.enabled:
178
+ continue
179
+
180
+ score = chunk_result.scores.get(condition.eval_name)
181
+ if score is None:
182
+ continue
183
+
184
+ # Track consecutive failures
185
+ key = condition.name
186
+ if self._check_threshold(score, condition.threshold, condition.comparison):
187
+ self._state.consecutive_failures[key] = (
188
+ self._state.consecutive_failures.get(key, 0) + 1
189
+ )
190
+ self._state.total_failures[key] = (
191
+ self._state.total_failures.get(key, 0) + 1
192
+ )
193
+
194
+ # Check if consecutive threshold met
195
+ if condition.check(score, self._state.consecutive_failures[key]):
196
+ self._state.triggered_conditions.append(condition.name)
197
+ return True, self._get_reason_for_condition(condition)
198
+ else:
199
+ # Reset consecutive count
200
+ self._state.consecutive_failures[key] = 0
201
+
202
+ # Check custom conditions
203
+ for check_fn in self._custom_checks:
204
+ reason = check_fn(chunk_result)
205
+ if reason is not None:
206
+ return True, reason
207
+
208
+ return False, EarlyStopReason.NONE
209
+
210
+ def _check_threshold(
211
+ self,
212
+ score: float,
213
+ threshold: float,
214
+ comparison: str,
215
+ ) -> bool:
216
+ """Check if score triggers threshold."""
217
+ if comparison == "below":
218
+ return score < threshold
219
+ else:
220
+ return score > threshold
221
+
222
+ def _get_reason_for_condition(
223
+ self,
224
+ condition: EarlyStopCondition,
225
+ ) -> EarlyStopReason:
226
+ """Get the appropriate stop reason for a condition."""
227
+ name_lower = condition.name.lower()
228
+ eval_lower = condition.eval_name.lower()
229
+
230
+ if "toxic" in name_lower or "toxic" in eval_lower:
231
+ return EarlyStopReason.TOXICITY
232
+ elif "safe" in name_lower or "safe" in eval_lower:
233
+ return EarlyStopReason.SAFETY
234
+ elif "pii" in name_lower or "pii" in eval_lower:
235
+ return EarlyStopReason.PII
236
+ elif "jailbreak" in name_lower or "jailbreak" in eval_lower:
237
+ return EarlyStopReason.JAILBREAK
238
+ else:
239
+ return EarlyStopReason.THRESHOLD
240
+
241
+ def reset(self) -> None:
242
+ """Reset policy state."""
243
+ self._state = PolicyState()
244
+
245
+ def enable_condition(self, name: str) -> None:
246
+ """Enable a condition by name."""
247
+ for condition in self._conditions:
248
+ if condition.name == name:
249
+ condition.enabled = True
250
+ break
251
+
252
+ def disable_condition(self, name: str) -> None:
253
+ """Disable a condition by name."""
254
+ for condition in self._conditions:
255
+ if condition.name == name:
256
+ condition.enabled = False
257
+ break
258
+
259
+ def get_stats(self) -> Dict[str, Any]:
260
+ """Get policy statistics."""
261
+ return {
262
+ "conditions": [c.to_dict() for c in self._conditions],
263
+ "consecutive_failures": dict(self._state.consecutive_failures),
264
+ "total_failures": dict(self._state.total_failures),
265
+ "triggered_conditions": list(self._state.triggered_conditions),
266
+ "custom_checks": len(self._custom_checks),
267
+ }
268
+
269
+ @classmethod
270
+ def default(cls) -> "EarlyStopPolicy":
271
+ """
272
+ Create a policy with sensible defaults.
273
+
274
+ Returns:
275
+ EarlyStopPolicy with toxicity and safety stops
276
+ """
277
+ policy = cls()
278
+ policy.add_toxicity_stop(threshold=0.7, consecutive=1)
279
+ policy.add_safety_stop(threshold=0.3, consecutive=1)
280
+ return policy
281
+
282
+ @classmethod
283
+ def strict(cls) -> "EarlyStopPolicy":
284
+ """
285
+ Create a strict policy for high-risk applications.
286
+
287
+ Returns:
288
+ EarlyStopPolicy with strict thresholds
289
+ """
290
+ policy = cls()
291
+ policy.add_toxicity_stop(threshold=0.5, consecutive=1)
292
+ policy.add_safety_stop(threshold=0.5, consecutive=1)
293
+ policy.add_quality_stop(threshold=0.4, consecutive=2)
294
+ return policy
295
+
296
+ @classmethod
297
+ def permissive(cls) -> "EarlyStopPolicy":
298
+ """
299
+ Create a permissive policy that only stops on severe issues.
300
+
301
+ Returns:
302
+ EarlyStopPolicy with high thresholds
303
+ """
304
+ policy = cls()
305
+ policy.add_toxicity_stop(threshold=0.9, consecutive=2)
306
+ policy.add_safety_stop(threshold=0.1, consecutive=2)
307
+ return policy
@@ -0,0 +1,368 @@
1
+ """Streaming-compatible scorer functions.
2
+
3
+ Provides lightweight evaluation functions optimized for streaming evaluation.
4
+ These scorers are designed to be fast and work with incremental text.
5
+ """
6
+
7
+ import re
8
+ from typing import Callable, List, Set
9
+
10
+
11
+ # Toxicity word lists (simplified for demonstration)
12
+ TOXIC_WORDS: Set[str] = {
13
+ "hate", "kill", "attack", "destroy", "violent", "threat",
14
+ "abuse", "harass", "racist", "sexist", "discriminate",
15
+ }
16
+
17
+ PROFANITY_WORDS: Set[str] = {
18
+ # Basic profanity patterns (simplified)
19
+ }
20
+
21
+ # PII patterns
22
+ PII_PATTERNS = {
23
+ "email": re.compile(r'\b[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Z|a-z]{2,}\b'),
24
+ "phone": re.compile(r'\b(?:\+?1[-.\s]?)?\(?[0-9]{3}\)?[-.\s]?[0-9]{3}[-.\s]?[0-9]{4}\b'),
25
+ "ssn": re.compile(r'\b\d{3}-\d{2}-\d{4}\b'),
26
+ "credit_card": re.compile(r'\b(?:\d{4}[-\s]?){3}\d{4}\b'),
27
+ "ip_address": re.compile(r'\b\d{1,3}\.\d{1,3}\.\d{1,3}\.\d{1,3}\b'),
28
+ }
29
+
30
+ # Jailbreak patterns
31
+ JAILBREAK_PATTERNS = [
32
+ re.compile(r'ignore\s+(?:all\s+)?(?:previous\s+)?instructions?', re.IGNORECASE),
33
+ re.compile(r'disregard\s+(?:all\s+)?(?:previous\s+)?instructions?', re.IGNORECASE),
34
+ re.compile(r'forget\s+(?:all\s+)?(?:previous\s+)?instructions?', re.IGNORECASE),
35
+ re.compile(r'you\s+are\s+now\s+(?:a\s+)?(?:different|new)', re.IGNORECASE),
36
+ re.compile(r'pretend\s+(?:you\s+are|to\s+be)', re.IGNORECASE),
37
+ re.compile(r'act\s+as\s+(?:if|though)', re.IGNORECASE),
38
+ ]
39
+
40
+
41
+ def toxicity_scorer(chunk: str, cumulative: str) -> float:
42
+ """
43
+ Score text for toxicity.
44
+
45
+ Returns a score from 0.0 (not toxic) to 1.0 (highly toxic).
46
+ Uses the cumulative text for better context.
47
+
48
+ Args:
49
+ chunk: Current chunk text
50
+ cumulative: All text so far
51
+
52
+ Returns:
53
+ Toxicity score (0.0 = safe, 1.0 = toxic)
54
+ """
55
+ text = cumulative.lower()
56
+ words = set(re.findall(r'\b\w+\b', text))
57
+
58
+ toxic_count = len(words.intersection(TOXIC_WORDS))
59
+ total_words = len(words) if words else 1
60
+
61
+ # Calculate toxicity ratio with diminishing returns
62
+ raw_score = toxic_count / max(total_words, 10)
63
+
64
+ # Scale to 0-1 range with sensitivity adjustment
65
+ score = min(1.0, raw_score * 5)
66
+
67
+ return score
68
+
69
+
70
+ def safety_scorer(chunk: str, cumulative: str) -> float:
71
+ """
72
+ Score text for general safety.
73
+
74
+ Returns a score from 0.0 (unsafe) to 1.0 (safe).
75
+ Higher is better (opposite of toxicity).
76
+
77
+ Args:
78
+ chunk: Current chunk text
79
+ cumulative: All text so far
80
+
81
+ Returns:
82
+ Safety score (0.0 = unsafe, 1.0 = safe)
83
+ """
84
+ # Inverse of toxicity
85
+ toxicity = toxicity_scorer(chunk, cumulative)
86
+ return 1.0 - toxicity
87
+
88
+
89
+ def pii_scorer(chunk: str, cumulative: str) -> float:
90
+ """
91
+ Score text for PII presence.
92
+
93
+ Returns a score from 0.0 (no PII) to 1.0 (contains PII).
94
+ Lower is better (no PII is good).
95
+
96
+ Args:
97
+ chunk: Current chunk text
98
+ cumulative: All text so far
99
+
100
+ Returns:
101
+ PII score (0.0 = no PII, 1.0 = contains PII)
102
+ """
103
+ text = cumulative
104
+ pii_found = 0
105
+
106
+ for pattern_name, pattern in PII_PATTERNS.items():
107
+ matches = pattern.findall(text)
108
+ pii_found += len(matches)
109
+
110
+ # Return 1.0 if any PII found, otherwise 0.0
111
+ # Could be weighted by severity in production
112
+ return min(1.0, pii_found * 0.5)
113
+
114
+
115
+ def jailbreak_scorer(chunk: str, cumulative: str) -> float:
116
+ """
117
+ Score text for jailbreak attempt patterns.
118
+
119
+ Returns a score from 0.0 (no jailbreak) to 1.0 (jailbreak detected).
120
+ Lower is better.
121
+
122
+ Args:
123
+ chunk: Current chunk text
124
+ cumulative: All text so far
125
+
126
+ Returns:
127
+ Jailbreak score (0.0 = safe, 1.0 = jailbreak detected)
128
+ """
129
+ text = cumulative
130
+
131
+ for pattern in JAILBREAK_PATTERNS:
132
+ if pattern.search(text):
133
+ return 1.0
134
+
135
+ return 0.0
136
+
137
+
138
+ def coherence_scorer(chunk: str, cumulative: str) -> float:
139
+ """
140
+ Score text for coherence.
141
+
142
+ A simple heuristic based on sentence structure.
143
+ Returns 1.0 for coherent text, lower for incoherent.
144
+
145
+ Args:
146
+ chunk: Current chunk text
147
+ cumulative: All text so far
148
+
149
+ Returns:
150
+ Coherence score (0.0 = incoherent, 1.0 = coherent)
151
+ """
152
+ text = cumulative.strip()
153
+
154
+ if not text:
155
+ return 1.0
156
+
157
+ # Count sentences
158
+ sentences = re.split(r'[.!?]+', text)
159
+ sentences = [s.strip() for s in sentences if s.strip()]
160
+
161
+ if not sentences:
162
+ return 0.5
163
+
164
+ # Check for basic coherence indicators
165
+ score = 1.0
166
+
167
+ # Penalize very short average sentence length
168
+ avg_words = sum(len(s.split()) for s in sentences) / len(sentences)
169
+ if avg_words < 3:
170
+ score -= 0.3
171
+
172
+ # Penalize excessive repetition
173
+ words = cumulative.lower().split()
174
+ if len(words) > 10:
175
+ unique_ratio = len(set(words)) / len(words)
176
+ if unique_ratio < 0.3:
177
+ score -= 0.4
178
+
179
+ # Penalize gibberish (high non-alpha ratio)
180
+ alpha_chars = sum(1 for c in text if c.isalpha() or c.isspace())
181
+ if len(text) > 0:
182
+ alpha_ratio = alpha_chars / len(text)
183
+ if alpha_ratio < 0.7:
184
+ score -= 0.3
185
+
186
+ return max(0.0, score)
187
+
188
+
189
+ def quality_scorer(chunk: str, cumulative: str) -> float:
190
+ """
191
+ Score text for overall quality.
192
+
193
+ Combines multiple heuristics for a quality assessment.
194
+
195
+ Args:
196
+ chunk: Current chunk text
197
+ cumulative: All text so far
198
+
199
+ Returns:
200
+ Quality score (0.0 = poor, 1.0 = high quality)
201
+ """
202
+ text = cumulative.strip()
203
+
204
+ if not text:
205
+ return 0.5
206
+
207
+ scores = []
208
+
209
+ # Coherence component
210
+ scores.append(coherence_scorer(chunk, cumulative))
211
+
212
+ # Length appropriateness (not too short, not repetitive)
213
+ words = text.split()
214
+ if len(words) > 5:
215
+ scores.append(0.8)
216
+ else:
217
+ scores.append(0.5)
218
+
219
+ # Punctuation presence
220
+ if re.search(r'[.!?,]', text):
221
+ scores.append(0.9)
222
+ else:
223
+ scores.append(0.6)
224
+
225
+ return sum(scores) / len(scores) if scores else 0.5
226
+
227
+
228
+ def create_keyword_scorer(
229
+ keywords: Set[str],
230
+ return_high_on_match: bool = True,
231
+ ) -> Callable[[str, str], float]:
232
+ """
233
+ Create a custom keyword-based scorer.
234
+
235
+ Args:
236
+ keywords: Set of keywords to detect
237
+ return_high_on_match: If True, returns high score on match
238
+
239
+ Returns:
240
+ Scorer function
241
+ """
242
+ keywords_lower = {k.lower() for k in keywords}
243
+
244
+ def scorer(chunk: str, cumulative: str) -> float:
245
+ text = cumulative.lower()
246
+ words = set(re.findall(r'\b\w+\b', text))
247
+
248
+ matches = len(words.intersection(keywords_lower))
249
+
250
+ if return_high_on_match:
251
+ return min(1.0, matches * 0.2)
252
+ else:
253
+ return max(0.0, 1.0 - matches * 0.2)
254
+
255
+ return scorer
256
+
257
+
258
+ def create_pattern_scorer(
259
+ patterns: List[re.Pattern],
260
+ return_high_on_match: bool = True,
261
+ ) -> Callable[[str, str], float]:
262
+ """
263
+ Create a custom regex pattern-based scorer.
264
+
265
+ Args:
266
+ patterns: List of compiled regex patterns
267
+ return_high_on_match: If True, returns high score on match
268
+
269
+ Returns:
270
+ Scorer function
271
+ """
272
+ def scorer(chunk: str, cumulative: str) -> float:
273
+ text = cumulative
274
+
275
+ for pattern in patterns:
276
+ if pattern.search(text):
277
+ return 1.0 if return_high_on_match else 0.0
278
+
279
+ return 0.0 if return_high_on_match else 1.0
280
+
281
+ return scorer
282
+
283
+
284
+ class CompositeScorer:
285
+ """
286
+ Combines multiple scorers with weights.
287
+
288
+ Example:
289
+ scorer = CompositeScorer()
290
+ scorer.add(toxicity_scorer, weight=2.0)
291
+ scorer.add(coherence_scorer, weight=1.0)
292
+
293
+ combined_score = scorer(chunk, cumulative)
294
+ """
295
+
296
+ def __init__(self):
297
+ """Initialize composite scorer."""
298
+ self._scorers: List[tuple] = [] # (scorer_fn, weight)
299
+
300
+ def add(
301
+ self,
302
+ scorer: Callable[[str, str], float],
303
+ weight: float = 1.0,
304
+ ) -> "CompositeScorer":
305
+ """
306
+ Add a scorer with weight.
307
+
308
+ Args:
309
+ scorer: Scorer function
310
+ weight: Weight for this scorer
311
+
312
+ Returns:
313
+ Self for chaining
314
+ """
315
+ self._scorers.append((scorer, weight))
316
+ return self
317
+
318
+ def __call__(self, chunk: str, cumulative: str) -> float:
319
+ """
320
+ Calculate weighted average of all scorers.
321
+
322
+ Args:
323
+ chunk: Current chunk text
324
+ cumulative: All text so far
325
+
326
+ Returns:
327
+ Weighted average score
328
+ """
329
+ if not self._scorers:
330
+ return 0.5
331
+
332
+ total_weight = sum(w for _, w in self._scorers)
333
+ weighted_sum = sum(
334
+ scorer(chunk, cumulative) * weight
335
+ for scorer, weight in self._scorers
336
+ )
337
+
338
+ return weighted_sum / total_weight if total_weight > 0 else 0.5
339
+
340
+
341
+ # Pre-configured composite scorers
342
+ def safety_composite_scorer(chunk: str, cumulative: str) -> float:
343
+ """
344
+ Composite scorer for overall safety.
345
+
346
+ Combines toxicity, PII, and jailbreak detection.
347
+ Returns 1.0 for safe, 0.0 for unsafe.
348
+ """
349
+ # Invert toxicity and PII scores (lower is better for them)
350
+ toxicity = 1.0 - toxicity_scorer(chunk, cumulative)
351
+ pii = 1.0 - pii_scorer(chunk, cumulative)
352
+ jailbreak = 1.0 - jailbreak_scorer(chunk, cumulative)
353
+
354
+ # Weighted combination (jailbreak is most critical)
355
+ return (toxicity * 0.3 + pii * 0.3 + jailbreak * 0.4)
356
+
357
+
358
+ def quality_composite_scorer(chunk: str, cumulative: str) -> float:
359
+ """
360
+ Composite scorer for overall quality.
361
+
362
+ Combines coherence and quality metrics.
363
+ Returns 1.0 for high quality, 0.0 for low quality.
364
+ """
365
+ coherence = coherence_scorer(chunk, cumulative)
366
+ quality = quality_scorer(chunk, cumulative)
367
+
368
+ return (coherence * 0.5 + quality * 0.5)