agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
fi/evals/protect.py ADDED
@@ -0,0 +1,671 @@
1
+ import re
2
+ import copy
3
+ import time
4
+ import warnings
5
+ from concurrent.futures import ThreadPoolExecutor, TimeoutError, as_completed
6
+ from typing import Dict, List, Optional, Tuple, Any
7
+ from urllib.parse import urlparse
8
+
9
+ from fi.api.types import HttpMethod, RequestConfig
10
+ from fi.evals.evaluator import EvalResponseHandler, Evaluator
11
+ from fi.evals.templates import (
12
+ DataPrivacyCompliance,
13
+ PromptInjection,
14
+ Toxicity,
15
+ BiasDetection,
16
+ )
17
+ from fi.evals.protect_input_adapter import ProtectInputAdapter
18
+ from fi.utils.routes import Routes
19
+ from fi.utils.utils import get_keys_from_env, get_base_url_from_env
20
+ from fi.utils.errors import InvalidAuthError, SDKException, InvalidValueType, MissingRequiredKey
21
+ from pydantic import ValidationError
22
+
23
+ PROTECT_FLASH_ID = "76"
24
+ SUPPORT_PROTECT_FLASH = True # feature toggle
25
+
26
+ class Protect:
27
+ """Client for protecting against unwanted content using various metrics"""
28
+
29
+ def __init__(self,
30
+ fi_api_key: Optional[str] = None,
31
+ fi_secret_key: Optional[str] = None,
32
+ fi_base_url: Optional[str] = None,
33
+ evaluator: Optional[Evaluator] = None):
34
+ """
35
+ Initialize Protect Class
36
+
37
+ Args:
38
+ evaluator: Instance of Evaluator to use for evaluations. If None, creates a new one.
39
+ """
40
+ env_api_key, env_secret_key = get_keys_from_env()
41
+ fi_api_key = env_api_key or fi_api_key
42
+ fi_secret_key = env_secret_key or fi_secret_key
43
+ fi_base_url = get_base_url_from_env() or fi_base_url
44
+ if not fi_api_key or not fi_secret_key:
45
+ raise InvalidAuthError("API key or secret key is missing for Protect initialization.")
46
+
47
+ self.evaluator = evaluator if evaluator is not None else Evaluator(
48
+ fi_api_key=fi_api_key,
49
+ fi_secret_key=fi_secret_key,
50
+ fi_base_url=fi_base_url
51
+ )
52
+
53
+ # Map metric names to their corresponding template classes
54
+ self.metric_map = {
55
+ "toxicity": Toxicity,
56
+ "bias_detection": BiasDetection,
57
+ "prompt_injection": PromptInjection,
58
+ "data_privacy_compliance": DataPrivacyCompliance,
59
+ # Deprecated aliases (still supported)
60
+ "content_moderation": Toxicity,
61
+ "security": PromptInjection,
62
+ }
63
+ self._deprecated_metrics = {
64
+ "content_moderation": "toxicity",
65
+ "security": "prompt_injection",
66
+ }
67
+
68
+ def _sanitize_reason(self, text: Optional[str]) -> Optional[str]:
69
+ """Ensure the traceback or server URL doesn't reach the end user."""
70
+ if not text:
71
+ return None
72
+ # Strip URLs, tracebacks, and noisy client errors
73
+ text = re.sub(r'https?://\S+', '[redacted]', text)
74
+ text = re.sub(r'(Traceback.*?$)', '[redacted]', text, flags=re.I|re.S)
75
+ text = re.sub(r'\b\d{3}\s+(Client|Server)\s+Error:.*', 'Request failed.', text, flags=re.I)
76
+ # Also redact bare hostnames/IPs (defense-in-depth)
77
+ text = re.sub(r'\b([a-z0-9-]+\.)+[a-z]{2,}\b', '[redacted]', text, flags=re.I)
78
+ text = re.sub(r'\b\d{1,3}(?:\.\d{1,3}){3}\b', '[redacted]', text)
79
+ return text.strip()
80
+
81
+ def _check_rule_sync(
82
+ self, rule: Dict, test_case: ProtectInputAdapter
83
+ ) -> Tuple[str, bool, Optional[str], Optional[str]]:
84
+ """
85
+ Synchronous version of rule checking
86
+
87
+ Returns:
88
+ Tuple[str, bool, Optional[str], Optional[str]]:
89
+ """
90
+ # thread_name = threading.current_thread().name
91
+ # start_time = time.time()
92
+ # print(f"Starting rule check for {rule['metric']} in thread {thread_name} at {start_time}")
93
+
94
+ template_class = self.metric_map[rule["metric"]]
95
+ if rule["metric"] == "Data Privacy":
96
+ template = template_class(
97
+ config={"call_type": "protect", "check_internet": False}
98
+ )
99
+ # template = template_class(config={"check_internet": False})
100
+ else:
101
+ template = template_class(config={"call_type": "protect"})
102
+ # template = template_class(config={})
103
+
104
+ payload = {
105
+ "inputs": [test_case.model_dump()],
106
+ "config": {
107
+ template.eval_id: template.config
108
+ },
109
+ }
110
+ # print("sending the request to: ", f"{self.evaluator._base_url}/{Routes.evaluate.value}")
111
+ # print("payload: ", payload["inputs"][0]["input"][:100])
112
+
113
+ try:
114
+ eval_result = self.evaluator.request(
115
+ config=RequestConfig(
116
+ method=HttpMethod.POST,
117
+ url=f"{self.evaluator._base_url}/{Routes.evaluate.value}",
118
+ json=payload,
119
+ timeout=3000
120
+ ),
121
+ response_handler=EvalResponseHandler,
122
+ )
123
+ except Exception:
124
+ err_msg = (
125
+ "We couldn't process this request. Check your input or your credit balance."
126
+ "If it keeps failing, contact support."
127
+ )
128
+ # Return a synthetic “failed” for this rule so the outer loop can present a clean error.
129
+ return rule["metric"], True, err_msg, None
130
+
131
+ # end_time = time.time()
132
+ # print(f"Completed rule check for {rule['metric']} in thread {thread_name} at {end_time} (took {end_time - start_time:.2f}s)")
133
+
134
+ reason_text: Optional[str] = None
135
+
136
+ if eval_result.eval_results:
137
+ result = eval_result.eval_results[0]
138
+ detected_values = [result.output]
139
+
140
+ should_trigger = False
141
+ if rule["type"] == "any":
142
+ should_trigger = any(
143
+ value in rule["contains"] for value in detected_values
144
+ )
145
+ elif rule["type"] == "all":
146
+ should_trigger = all(
147
+ value in rule["contains"] for value in detected_values
148
+ )
149
+
150
+ if should_trigger:
151
+ if rule["_internal_reason_flag"]:
152
+ # message = rule['action'] + f' Reason: {result.reason}'
153
+ message = rule["action"]
154
+ reason_text = self._sanitize_reason(result.reason) if rule["_internal_reason_flag"] else None
155
+ else:
156
+ message = rule["action"]
157
+ return rule["metric"], True, message, reason_text
158
+
159
+ return rule["metric"], False, None, None
160
+
161
+ def _process_rules_batch(
162
+ self, rules: List[Dict], test_case: ProtectInputAdapter, remaining_time: float
163
+ ) -> Tuple[List[str], List[str], List[str], List[str], List[str]]:
164
+ """
165
+ Process a batch of rules in parallel
166
+
167
+ Args:
168
+ rules: List of rules to process
169
+ test_case: Test case to evaluate
170
+ remaining_time: Time remaining for processing
171
+
172
+ Returns:
173
+ Tuple[List[str], List[str], List[str], List[str], List[str]]:
174
+ (failure_messages, completed_rules, uncompleted_rules, failure_reasons, failed_rule)
175
+ """
176
+ # print(f"\nProcessing batch of {len(rules)} rules")
177
+ # batch_start = time.time()
178
+
179
+ completed_rules = []
180
+ uncompleted_rules = [rule["metric"] for rule in rules]
181
+ failure_messages = []
182
+ failure_reasons = []
183
+ failed_rule = []
184
+
185
+ with ThreadPoolExecutor(max_workers=5) as executor:
186
+ # Submit all rules to the thread pool
187
+ future_to_rule = {
188
+ executor.submit(self._check_rule_sync, rule, test_case): rule["metric"]
189
+ for rule in rules
190
+ }
191
+
192
+ try:
193
+ # Wait for futures to complete with timeout
194
+ for future in as_completed(future_to_rule, timeout=remaining_time):
195
+ rule_name = future_to_rule[future]
196
+ try:
197
+ metric, triggered, message, reason_text = future.result()
198
+ # Update tracking lists
199
+ completed_rules.append(metric)
200
+ if rule_name in uncompleted_rules:
201
+ uncompleted_rules.remove(rule_name)
202
+
203
+ if triggered:
204
+ failure_messages.append(message)
205
+ if reason_text:
206
+ failure_reasons.append(reason_text)
207
+ failed_rule.append(rule_name)
208
+ # Cancel remaining futures if a rule fails
209
+ for f_key, f_val in future_to_rule.items():
210
+ if not f_key.done():
211
+ f_key.cancel()
212
+
213
+ except Exception:
214
+ if rule_name in uncompleted_rules:
215
+ # uncompleted_rules.remove(rule_name) # Errored rule should remain uncompleted
216
+ pass
217
+
218
+ except TimeoutError:
219
+ # print(
220
+ # f"Timeout reached. {len(completed_rules)} rules completed, "
221
+ # f"{len(uncompleted_rules)} rules incomplete"
222
+ # )
223
+ all_submitted_rules = [r["metric"] for r in rules]
224
+ uncompleted_rules = [r for r in all_submitted_rules if r not in completed_rules]
225
+
226
+ # batch_end = time.time()
227
+ # print(f"Batch processing completed in {batch_end - batch_start:.2f}s\n")
228
+
229
+ return (
230
+ failure_messages,
231
+ completed_rules,
232
+ uncompleted_rules,
233
+ failure_reasons,
234
+ failed_rule,
235
+ )
236
+
237
+ def _is_url(self, text: str) -> bool:
238
+ """
239
+ Check if the input text is a URL or URL-like string.
240
+
241
+ Args:
242
+ text: String to check
243
+
244
+ Returns:
245
+ bool: True if input appears to be a URL, False otherwise
246
+ """
247
+ # Check if it's an explicit URL with a scheme
248
+ parsed_url = urlparse(text)
249
+ if parsed_url.scheme in ['http', 'https']:
250
+ return True
251
+
252
+ # Check for URL-like patterns without scheme
253
+ text_lower = text.lower()
254
+ # Check for common TLDs
255
+ common_tlds = ['.com', '.org', '.net', '.edu', '.gov', '.io', '.co']
256
+ has_tld = any(tld in text_lower for tld in common_tlds)
257
+
258
+ # Check for patterns like "www." at the beginning
259
+ starts_with_www = text_lower.startswith('www.')
260
+
261
+ # Check for domain-like pattern (example.com)
262
+ has_domain_pattern = '.' in text_lower and not text_lower.startswith('.') and not text_lower.endswith('.')
263
+
264
+ return (has_tld and has_domain_pattern) or starts_with_www
265
+
266
+ def _is_only_url(self, text: str) -> bool:
267
+ """
268
+ Check if the input text is solely a URL without additional content.
269
+
270
+ Args:
271
+ text: String to check
272
+
273
+ Returns:
274
+ bool: True if the entire input appears to be just a URL, False otherwise
275
+ """
276
+ # Remove whitespace for checking
277
+ text = text.strip()
278
+
279
+ # If the string contains spaces, it's not only a URL
280
+ if ' ' in text:
281
+ return False
282
+
283
+ # Check if what remains is a URL
284
+ return self._is_url(text)
285
+
286
+ def _format_adapter_error(self, ve: ValidationError) -> str:
287
+ """
288
+ Convert Pydantic validation errors into a single, user-friendly line.
289
+ No stack traces, no class names, no internal URLs.
290
+ """
291
+ # Default fallback:
292
+ friendly = (
293
+ "We couldn't read that input. Please use text, a direct media URL, "
294
+ "or a supported file type (MP3/WAV for audio; JPG/PNG/WebP/GIF/BMP/TIFF/SVG for images)."
295
+ )
296
+
297
+ try:
298
+ for err in ve.errors():
299
+ msg = (err.get("msg") or "").lower()
300
+
301
+ # Specific: unsupported local file type (e.g., .ogg)
302
+ if "unsupported local file type" in msg:
303
+ return (
304
+ "Unsupported file type. Supported audio: MP3, WAV. "
305
+ "Supported image: JPG, PNG, WebP, GIF, BMP, TIFF, SVG."
306
+ )
307
+
308
+ # Specific: data: URI with non-media
309
+ if "unsupported data uri mime" in msg:
310
+ return "Only audio/* or image/* data: URIs are supported."
311
+
312
+ # Specific: empty/whitespace
313
+ if "input cannot be empty" in msg:
314
+ return "Input cannot be empty."
315
+
316
+ # Specific: looks like a preview page not a raw file
317
+ if "preview page, not a direct file" in msg:
318
+ return (
319
+ "This link looks like a preview page. Please use a direct download URL "
320
+ "(e.g., raw.githubusercontent.com for GitHub; export=download for Google Drive; "
321
+ "dl.dropboxusercontent.com for Dropbox)."
322
+ )
323
+ except Exception:
324
+ pass
325
+
326
+ return friendly
327
+
328
+ def protect(
329
+ self,
330
+ inputs: str,
331
+ protect_rules: Optional[List[Dict]] = None,
332
+ action: str = "Response cannot be generated as the input fails the checks",
333
+ reason: bool = False,
334
+ timeout: float = 30000, #milliseconds
335
+ use_flash: bool = False,
336
+ ) -> Dict[str, Any]:
337
+ """
338
+ Evaluate input strings against protection rules
339
+
340
+ Args:
341
+ inputs: Text or list of texts to check for harmful content
342
+ timeout: Time limit for evaluation in milliseconds (default: 30000)
343
+ protect_rules: Rules to check against. Each rule needs:
344
+ metric: What to check (e.g. 'content_moderation', 'bias_detection')
345
+ contains: Values to look for
346
+ type: 'any' or 'all' matching required
347
+ action: Message to show if rule fails
348
+ reason: Include explanation in message (optional)
349
+ use_flash: Use fast binary classification instead of detailed rules
350
+
351
+ Returns:
352
+ Dictionary containing:
353
+ status: "passed" or "failed"
354
+ completed_rules: List of rules that were checked
355
+ uncompleted_rules: List of rules that couldn't be checked
356
+ failed_rule: The rule that triggered the failure (if any)
357
+ messages: The action message for the failed rule or the input if passed
358
+ reasons: The reason for failure or "All checks passed"
359
+ time_taken: Elapsed time for the evaluation
360
+
361
+ Raises:
362
+ ValueError: If inputs or protect_rules don't match the required structure
363
+ TypeError: If inputs contains non-string objects
364
+ """
365
+
366
+
367
+ timeout_seconds = timeout / 1000.0
368
+
369
+ if protect_rules is None:
370
+ protect_rules = []
371
+
372
+ # If the caller asked for Flash but we don’t support it yet, fall back.
373
+ if use_flash and not SUPPORT_PROTECT_FLASH:
374
+ # Provide a sensible default so behavior is still helpful.
375
+ if not protect_rules:
376
+ protect_rules = [{"metric": "content_moderation"}]
377
+ use_flash = False # force normal path
378
+
379
+ # When using ProtectFlash and no protect_rules provided, create default rules
380
+ if use_flash and not protect_rules:
381
+ protect_rules = [{"metric": "content_moderation"}]
382
+ elif use_flash and protect_rules:
383
+ print("Note: When using ProtectFlash, Rules are not considered as it performs binary harmful/not harmful classification only.")
384
+
385
+ # Ensure protect_rules is a list
386
+ if protect_rules is None:
387
+ protect_rules = []
388
+
389
+ protect_rules_copy = copy.deepcopy(protect_rules)
390
+
391
+ # Validate inputs
392
+ if inputs is None:
393
+ raise InvalidValueType(value_name="inputs", value=inputs, correct_type="string or list of strings")
394
+
395
+ # This check can be more specific if we only expect str initially that gets converted
396
+ if not isinstance(inputs, (str, list)):
397
+ raise InvalidValueType(value_name="inputs", value=inputs, correct_type="string or list of strings")
398
+
399
+ # Convert single string to list for uniform processing
400
+ if isinstance(inputs, str):
401
+ inputs_list = [inputs]
402
+ else:
403
+ inputs_list = inputs # Already a list
404
+
405
+ if not inputs_list: # Check after potential conversion
406
+ raise InvalidValueType(value_name="inputs", value=inputs_list, correct_type="non-empty string or non-empty list of strings")
407
+
408
+ # Validate each input is a non-empty string
409
+ for i, input_text in enumerate(inputs_list):
410
+ if not isinstance(input_text, str):
411
+ raise InvalidValueType(
412
+ value_name=f"input at index {i}",
413
+ value=input_text,
414
+ correct_type="string"
415
+ )
416
+ if not input_text.strip():
417
+ raise InvalidValueType(
418
+ value_name=f"input at index {i}",
419
+ value=input_text,
420
+ correct_type="non-empty string or string with non-whitespace characters"
421
+ )
422
+
423
+ # If using ProtectFlash, we can use a simpler approach by directly calling the API with protect_flash=True
424
+ if use_flash and SUPPORT_PROTECT_FLASH:
425
+ # Create a test case with appropriate payload
426
+ test_case = ProtectInputAdapter(input=inputs, call_type="protect")
427
+
428
+ # Prepare the protect API call with protect_flash flag
429
+ template_class = self.metric_map[protect_rules_copy[0]["metric"]]
430
+ template = template_class(config={"call_type": "protect"})
431
+
432
+ # Ensure action is set in the rule for consistency with standard protect
433
+ # This allows the user to provide action either as a parameter or in the rule itself
434
+ if "action" not in protect_rules_copy[0]:
435
+ protect_rules_copy[0]["action"] = action
436
+
437
+ # Call the evaluator with protect_flash=True
438
+ # Custom payload for ProtectFlash that includes the protect_flash flag
439
+ payload = {
440
+ "inputs": [test_case.model_dump()],
441
+ "config": {
442
+ PROTECT_FLASH_ID: template.config
443
+ },
444
+ "protect_flash": True # This is the key flag that enables ProtectFlash
445
+ }
446
+
447
+ # Make a direct request using the evaluator's request method
448
+ response = self.evaluator.request(
449
+ config=RequestConfig(
450
+ method=HttpMethod.POST,
451
+ url=f"{self.evaluator._base_url}/{Routes.evaluate.value}",
452
+ json=payload,
453
+ timeout=timeout / 1000 or self.evaluator._default_timeout,
454
+ ),
455
+ response_handler=EvalResponseHandler,
456
+ )
457
+ # Process the response
458
+ if hasattr(response, "eval_results") and response.eval_results:
459
+ result = response.eval_results[0]
460
+ # result.output can be:
461
+ # - A float score (0.0 = harmful, 1.0 = safe) from ProtectFlash metrics
462
+ # - A string "Failed" or "Passed" from other evaluators
463
+ if isinstance(result.output, (int, float)):
464
+ is_harmful = result.output == 0.0
465
+ else:
466
+ is_harmful = result.output == "Failed"
467
+ elapsed_time = result.runtime / 1000 if result.runtime else 0
468
+
469
+ ans = {
470
+ "status": "failed" if is_harmful else "passed",
471
+ "completed_rules": ["ProtectFlash"], # Use ProtectFlash instead of rule metric
472
+ "uncompleted_rules": [],
473
+ "failed_rule": "ProtectFlash" if is_harmful else None, # Use ProtectFlash instead of rule metric
474
+ "messages": protect_rules_copy[0]["action"] if is_harmful else inputs[0],
475
+ "reasons": ["Content detected as harmful." if is_harmful else "All checks passed"],
476
+ "time_taken": elapsed_time,
477
+ }
478
+ return ans
479
+ else:
480
+ # Return a default response if no results
481
+ return {
482
+ "status": "error",
483
+ "messages": "Evaluation failed",
484
+ "completed_rules": [],
485
+ "uncompleted_rules": ["ProtectFlash"],
486
+ "failed_rule": None,
487
+ "reasons": ["No evaluation results returned"],
488
+ "time_taken": 0,
489
+ }
490
+
491
+
492
+ # Original implementation for standard Protect (non-flash)
493
+ # Convert inputs to MLLMTestCase instances with call_type="protect"
494
+ try:
495
+ test_cases = [ProtectInputAdapter(input=input_text, call_type="protect") for input_text in inputs_list]
496
+ except ValidationError as ve:
497
+ msg = self._format_adapter_error(ve)
498
+ return {
499
+ "status": "failed",
500
+ "completed_rules": [],
501
+ "uncompleted_rules": [r["metric"] for r in protect_rules_copy],
502
+ "failed_rule": [],
503
+ "messages": msg,
504
+ "reasons": ["No evaluation results returned"],
505
+ "time_taken": 0,
506
+ }
507
+ # Validate protect_rules_copy
508
+ if not isinstance(protect_rules_copy, list):
509
+ raise InvalidValueType(value_name="protect_rules", value=protect_rules_copy, correct_type="list")
510
+
511
+ if not protect_rules_copy:
512
+ raise InvalidValueType(value_name="protect_rules", value=protect_rules_copy, correct_type="non-empty list")
513
+
514
+ valid_metrics = set(self.metric_map.keys())
515
+ valid_types = {"any", "all"}
516
+
517
+ for i, rule in enumerate(protect_rules_copy):
518
+
519
+ if not isinstance(rule, dict):
520
+ raise InvalidValueType(value_name=f"Rule at index {i}", value=rule, correct_type="dictionary")
521
+
522
+ # Check required keys
523
+ required_keys = {"metric"}
524
+ missing_keys = required_keys - set(rule.keys())
525
+ if missing_keys:
526
+ # Using MissingRequiredKey from our errors module
527
+ raise MissingRequiredKey(field_name=f"Rule at index {i}", missing_key=', '.join(missing_keys))
528
+
529
+ # Validate metric name first, as other validations might depend on it
530
+ if rule["metric"] not in valid_metrics:
531
+ raise InvalidValueType(
532
+ value_name=f"metric in Rule at index {i}",
533
+ value=rule["metric"],
534
+ correct_type=f"one of {list(valid_metrics)}"
535
+ )
536
+
537
+ # Warn about deprecated metric names
538
+ if rule["metric"] in self._deprecated_metrics:
539
+ new_name = self._deprecated_metrics[rule["metric"]]
540
+ warnings.warn(
541
+ f'Protect metric "{rule["metric"]}" is deprecated and will be '
542
+ f'removed in a future release. Please use "{new_name}" instead.',
543
+ FutureWarning,
544
+ stacklevel=2,
545
+ )
546
+
547
+ is_tone_metric = rule["metric"] == "Tone"
548
+
549
+ if is_tone_metric:
550
+ if "contains" not in rule:
551
+ raise MissingRequiredKey(field_name=f"Rule for Tone metric at index {i}", missing_key="contains")
552
+ if not isinstance(rule["contains"], list):
553
+ raise InvalidValueType(value_name=f"'contains' in Tone rule at index {i}", value=rule["contains"], correct_type="list")
554
+ if not rule["contains"]:
555
+ raise InvalidValueType(value_name=f"'contains' in Tone rule at index {i}", value=rule["contains"], correct_type="non-empty list")
556
+
557
+ # Type for Tone metric
558
+ if "type" not in rule:
559
+ rule["type"] = "any" # Default if not present
560
+ elif rule["type"] not in valid_types:
561
+ raise InvalidValueType(
562
+ value_name=f"'type' in Tone rule at index {i}",
563
+ value=rule["type"],
564
+ correct_type=f"one of {valid_types}"
565
+ )
566
+ else: # For non-Tone metrics
567
+ if "contains" in rule:
568
+ # This indicates an invalid configuration for a non-Tone metric
569
+ raise SDKException(f"'contains' should not be specified for {rule['metric']} metric at index {i}. Provide it only for 'Tone' metric.")
570
+ if "type" in rule:
571
+ raise SDKException(f"'type' should not be specified for {rule['metric']} metric at index {i}. Provide it only for 'Tone' metric.")
572
+
573
+ # Set default values for internal processing of non-Tone metrics
574
+ rule["contains"] = ["Failed"] # Predefined internal value to check against for non-Tone metrics
575
+ rule["type"] = "any" # Default type for non-Tone metrics
576
+
577
+ # Validate action
578
+ if "action" not in rule:
579
+ rule["action"] = action # Default action if not specified
580
+
581
+ # 'reason' should not be in the input rule, it's a parameter to the protect method itself.
582
+ if "reason" in rule:
583
+ raise InvalidValueType(value_name=f"key in rule at index {i}", value="reason", correct_type="not to be part of the rule, it is a global parameter")
584
+ # Set the global reason for this rule processing from the method's parameter
585
+ rule["_internal_reason_flag"] = reason # Use a different key to avoid conflict
586
+
587
+ # results = []
588
+ BATCH_SIZE = 5 # Maximum number of concurrent rule checks
589
+ if len(protect_rules_copy) < BATCH_SIZE:
590
+ BATCH_SIZE = len(protect_rules_copy)
591
+
592
+ # total_timeout = timeout # Original line, timeout is in ms
593
+ total_timeout_for_processing_seconds = timeout_seconds
594
+ start_time = time.time()
595
+
596
+ all_failure_messages = []
597
+ all_completed_rules = []
598
+ all_uncompleted_rules = []
599
+ all_failure_reasons = []
600
+ # try:
601
+ bool_check_fail = False
602
+ for test_case in test_cases:
603
+ for i in range(0, len(protect_rules_copy), BATCH_SIZE):
604
+ # Calculate remaining time
605
+ elapsed_time = time.time() - start_time # This is in seconds
606
+ # remaining_time = max(0, total_timeout - elapsed_time) # BUG: total_timeout was ms
607
+ remaining_time_seconds = max(0, total_timeout_for_processing_seconds - elapsed_time)
608
+
609
+ if remaining_time_seconds <= 0:
610
+ # Add remaining rules to uncompleted list
611
+ remaining_rules = [
612
+ rule["metric"] for rule in protect_rules_copy[i:]
613
+ ]
614
+ all_uncompleted_rules.extend(remaining_rules)
615
+ break
616
+
617
+ rules_batch = protect_rules_copy[i : i + BATCH_SIZE]
618
+ (
619
+ messages,
620
+ completed,
621
+ uncompleted,
622
+ failure_reasons,
623
+ failed_rule,
624
+ ) = self._process_rules_batch(rules_batch, test_case, remaining_time_seconds)
625
+
626
+ all_completed_rules.extend(completed)
627
+ all_uncompleted_rules.extend(uncompleted)
628
+ all_failure_reasons.extend(failure_reasons)
629
+ if messages:
630
+ all_failure_messages.extend(messages)
631
+ bool_check_fail = True
632
+ break
633
+
634
+ final_processing_duration_seconds = time.time() - start_time
635
+
636
+ ans = {
637
+ "status": "failed" if all_failure_messages else "passed",
638
+ "completed_rules": all_completed_rules,
639
+ "uncompleted_rules": all_uncompleted_rules,
640
+ "failed_rule": failed_rule,
641
+ "messages": (
642
+ all_failure_messages[0] if all_failure_messages else "All checks passed"
643
+ ),
644
+ "reasons": (
645
+ all_failure_reasons if all_failure_reasons else ["All checks passed"]
646
+ ),
647
+ "time_taken": final_processing_duration_seconds, # Use final calculated duration in seconds
648
+ }
649
+
650
+ if len(ans["uncompleted_rules"]) == len(protect_rules_copy):
651
+ ans["reason"] = "No checks completed"
652
+
653
+ if bool_check_fail:
654
+ ans["status"] = "failed"
655
+ else:
656
+ ans["status"] = "passed"
657
+ # ans['messages'] = inputs
658
+
659
+ if ans["status"] == "passed":
660
+ ans["messages"] = inputs_list[0]
661
+
662
+ return ans
663
+
664
+ def protect(
665
+ inputs,
666
+ protect_rules,
667
+ action="Response cannot be generated as the input fails the checks",
668
+ reason=False,
669
+ timeout=30000,
670
+ ):
671
+ return Protect().protect(inputs, protect_rules, action, reason, timeout)