agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,449 @@
1
+ import optuna
2
+ import logging
3
+ import random
4
+ import json
5
+ import re
6
+ import time
7
+ from typing import List, Dict, Any, Optional, Callable
8
+ from ..base.base_optimizer import BaseOptimizer
9
+ from ..types import OptimizationResult, IterationHistory, EvaluationResult
10
+ from ..datamappers import BasicDataMapper
11
+ from ..generators.litellm import LiteLLMGenerator
12
+ from ..base.evaluator import Evaluator
13
+ from ..utils.early_stopping import EarlyStoppingConfig, EarlyStoppingChecker
14
+
15
+
16
+ TEACHER_SYSTEM_PROMPT = (
17
+ """
18
+ You are an expert prompt engineer with deep knowledge of few-shot learning and template design. Your task is to analyze a sample of dataset items and create an optimal Python .format() string template for few-shot examples.
19
+
20
+ ANALYSIS REQUIREMENTS:
21
+ 1. Examine the structure and content of the provided dataset examples
22
+ 2. Identify all available field names/keys in the examples
23
+ 3. Determine which fields represent inputs vs. expected outputs
24
+ 4. Design a template that clearly demonstrates the input-output relationship
25
+
26
+ TEMPLATE DESIGN PRINCIPLES:
27
+ - Use ONLY field names that actually exist in the provided examples
28
+ - Include both input and output fields to enable effective few-shot learning
29
+ - Create clear, readable formatting that helps models understand the pattern
30
+ - Use descriptive labels (e.g., "Input:", "Output:", "Question:", "Answer:")
31
+ - Ensure the template is concise yet informative
32
+ - Maintain consistent formatting across examples
33
+
34
+ OUTPUT FORMAT:
35
+ Return ONLY a valid JSON object with this exact structure:
36
+ {
37
+ "example_template": "your_template_string_here"
38
+ }
39
+
40
+ The template string must:
41
+ - Use Python .format() syntax with curly braces for field substitution
42
+ - Include clear labels for input and output sections
43
+ - Be ready to use without any modifications
44
+ - Work for all examples in the dataset
45
+
46
+ Example of a well-formed template:
47
+ "Question: {question}\nAnswer: {answer}"
48
+ or
49
+ "Prompt: {prompt}\nExpected Response: {response}\n---"
50
+
51
+ DO NOT include any explanations, comments, or additional text - only the JSON object.
52
+ """
53
+ ).strip()
54
+
55
+
56
+ class BayesianSearchOptimizer(BaseOptimizer):
57
+ """
58
+ An optimizer that uses Bayesian optimization (via Optuna) to find the
59
+ best prompt by intelligently selecting few-shot examples.
60
+ """
61
+
62
+ def __init__(
63
+ self,
64
+ # Few-shot search space
65
+ min_examples: int = 2,
66
+ max_examples: int = 8,
67
+ allow_repeats: bool = False,
68
+ fixed_example_indices: Optional[List[int]] = None,
69
+ # Trials and randomness
70
+ n_trials: int = 10,
71
+ seed: int = 42,
72
+ # Inference/generation config
73
+ inference_model_name: str = "gpt-4o-mini",
74
+ inference_model_kwargs: Optional[Dict[str, Any]] = None,
75
+ # Example formatting and prompt construction
76
+ example_template: Optional[str] = None,
77
+ example_template_fields: Optional[List[str]] = None,
78
+ field_aliases: Optional[Dict[str, str]] = None,
79
+ example_separator: str = "\n",
80
+ few_shot_position: str = "append", # "prepend" | "append"
81
+ prompt_builder: Optional[Callable[[str, List[str]], str]] = None,
82
+ example_formatter: Optional[Callable[[Dict[str, Any]], str]] = None,
83
+ few_shot_title: Optional[str] = None,
84
+ # Teacher-guided template inference (optional)
85
+ infer_example_template_via_teacher: bool = False,
86
+ teacher_model_name: str = "gpt-5",
87
+ teacher_model_kwargs: Optional[Dict[str, Any]] = None,
88
+ template_infer_n_samples: int = 8,
89
+ teacher_system_prompt: str = TEACHER_SYSTEM_PROMPT,
90
+ teacher_infer_max_retries: int = 2,
91
+ teacher_infer_retry_sleep: float = 0.5,
92
+ # Evaluation controls
93
+ eval_subset_size: Optional[int] = None,
94
+ eval_subset_strategy: str = "random", # "random" | "first" | "all"
95
+ score_aggregator: Optional[Callable[[List[EvaluationResult]], float]] = None,
96
+ # Optuna controls
97
+ sampler: Optional[optuna.samplers.BaseSampler] = None,
98
+ pruner: Optional[optuna.pruners.BasePruner] = None,
99
+ direction: str = "maximize",
100
+ storage: Optional[str] = None,
101
+ study_name: Optional[str] = None,
102
+ ):
103
+ # Search space
104
+ self.min_examples = min_examples
105
+ self.max_examples = max_examples
106
+ self.allow_repeats = allow_repeats
107
+ self.fixed_example_indices = fixed_example_indices or []
108
+ # Trials and randomness
109
+ self.n_trials = n_trials
110
+ self.seed = seed
111
+ # Inference/generation
112
+ self.inference_model_name = inference_model_name
113
+ self.inference_model_kwargs = inference_model_kwargs or {}
114
+ # Formatting/building
115
+ self.example_template = example_template
116
+ self.example_template_fields = example_template_fields
117
+ self.field_aliases = field_aliases or {}
118
+ self.example_separator = example_separator
119
+ self.few_shot_position = few_shot_position
120
+ self.prompt_builder = prompt_builder
121
+ self.example_formatter = example_formatter
122
+ self.few_shot_title = few_shot_title
123
+ # Teacher-guided template inference
124
+ self.infer_example_template_via_teacher = infer_example_template_via_teacher
125
+ self.teacher_model_name = teacher_model_name
126
+ # default kwargs for gpt-5 style models
127
+ default_teacher_kwargs: Dict[str, Any] = {
128
+ "temperature": 1.0,
129
+ "max_tokens": 16000,
130
+ }
131
+ self.teacher_model_kwargs = {
132
+ **default_teacher_kwargs,
133
+ **(teacher_model_kwargs or {}),
134
+ }
135
+ self.template_infer_n_samples = template_infer_n_samples
136
+ self.teacher_system_prompt = teacher_system_prompt
137
+ self.teacher_infer_max_retries = max(0, int(teacher_infer_max_retries))
138
+ self.teacher_infer_retry_sleep = max(0.0, float(teacher_infer_retry_sleep))
139
+ # Evaluation
140
+ self.eval_subset_size = eval_subset_size
141
+ self.eval_subset_strategy = eval_subset_strategy
142
+ self.score_aggregator = score_aggregator or self._default_score_aggregator
143
+ # Optuna
144
+ self.sampler = sampler or optuna.samplers.TPESampler(seed=self.seed)
145
+ self.pruner = pruner
146
+ self.direction = direction
147
+ self.storage = storage
148
+ self.study_name = study_name
149
+ # runtime state
150
+ self._runtime_example_template: Optional[str] = None
151
+
152
+ def optimize(
153
+ self,
154
+ evaluator: Evaluator,
155
+ data_mapper: BasicDataMapper,
156
+ dataset: List[Dict[str, Any]],
157
+ initial_prompts: List[str],
158
+ early_stopping: Optional[EarlyStoppingConfig] = None,
159
+ **kwargs: Any,
160
+ ) -> OptimizationResult:
161
+ logging.info("--- Starting Bayesian Search Optimization ---")
162
+
163
+ # Initialize early stopping checker
164
+ checker = None
165
+ if early_stopping and early_stopping.is_enabled():
166
+ checker = EarlyStoppingChecker(early_stopping)
167
+ logging.info(f"Early stopping enabled: {early_stopping}")
168
+
169
+ if not initial_prompts:
170
+ raise ValueError("Initial prompts list cannot be empty.")
171
+
172
+ initial_prompt = initial_prompts[0]
173
+ history: List[IterationHistory] = []
174
+
175
+ # Optionally infer the example template via a teacher model from a sample of the dataset
176
+ self._runtime_example_template = None
177
+ if self.infer_example_template_via_teacher:
178
+ try:
179
+ self._runtime_example_template = self._infer_example_template(dataset)
180
+ logging.info(
181
+ f"Inferred example template via teacher model: \n {self._runtime_example_template}"
182
+ )
183
+ except Exception as e:
184
+ logging.warning(f"Falling back to default example_template. Error: {e}")
185
+ self._runtime_example_template = None
186
+
187
+ def objective(trial: optuna.Trial) -> float:
188
+ # Suggest number of few-shot examples
189
+ n_examples = trial.suggest_int(
190
+ "n_examples", self.min_examples, self.max_examples
191
+ )
192
+
193
+ # Use a single seed to derive indices, avoiding dynamic value spaces in Optuna
194
+ example_seed = trial.suggest_int("example_seed", 0, 2_000_000_000)
195
+ rng = random.Random(example_seed)
196
+
197
+ # Honor fixed indices first
198
+ selected_indices: List[int] = list(self.fixed_example_indices)
199
+ remaining_needed = max(0, n_examples - len(selected_indices))
200
+
201
+ if remaining_needed > 0:
202
+ if self.allow_repeats:
203
+ # Repeats allowed: sample with replacement
204
+ more = [
205
+ rng.randrange(len(dataset)) for _ in range(remaining_needed)
206
+ ]
207
+ selected_indices.extend(more)
208
+ else:
209
+ # Unique sampling: sample without replacement from remaining pool
210
+ pool = [
211
+ i for i in range(len(dataset)) if i not in set(selected_indices)
212
+ ]
213
+ take = min(remaining_needed, len(pool))
214
+ selected_indices.extend(rng.sample(pool, take))
215
+
216
+ # Format the selected examples for few-shot
217
+ demo_examples = [dataset[i] for i in selected_indices]
218
+ example_strings = [self._format_example(ex) for ex in demo_examples]
219
+ few_shot_block = self._build_few_shot_block(example_strings)
220
+
221
+ # Build the full prompt
222
+ full_prompt = self._build_prompt(initial_prompt, few_shot_block)
223
+
224
+ # Score the prompt
225
+ iteration_history = self._score_prompt(
226
+ full_prompt, evaluator, data_mapper, dataset
227
+ )
228
+
229
+ if not iteration_history:
230
+ trial.set_user_attr("prompt", full_prompt)
231
+ return 0.0
232
+
233
+ history.append(iteration_history)
234
+ avg_score = iteration_history.average_score
235
+ trial.set_user_attr("prompt", full_prompt)
236
+ logging.info(
237
+ f"Trial {trial.number}: Score={avg_score:.4f}, Num Examples={len(selected_indices)}"
238
+ )
239
+
240
+ # Check early stopping
241
+ if checker:
242
+ eval_size = len(self._select_eval_subset(dataset))
243
+ if checker.should_stop(avg_score, eval_size):
244
+ logging.info(
245
+ f"Early stopping triggered: {checker.get_state()['stop_reason']}"
246
+ )
247
+ trial.study.stop()
248
+
249
+ return avg_score
250
+
251
+ study = optuna.create_study(
252
+ direction=self.direction,
253
+ sampler=self.sampler,
254
+ pruner=self.pruner,
255
+ storage=self.storage,
256
+ study_name=self.study_name,
257
+ load_if_exists=bool(self.storage and self.study_name),
258
+ )
259
+
260
+ try:
261
+ study.optimize(objective, n_trials=self.n_trials)
262
+ except Exception as e:
263
+ logging.info(f"Optimization stopped: {e}")
264
+
265
+ # Check if any trials completed before accessing best_trial
266
+ if not history:
267
+ raise RuntimeError(
268
+ "Optimization stopped before any trials completed successfully"
269
+ )
270
+
271
+ best_prompt = study.best_trial.user_attrs.get("prompt", initial_prompt)
272
+ best_generator = LiteLLMGenerator(self.inference_model_name, best_prompt)
273
+
274
+ # Build result with early stopping metadata
275
+ return OptimizationResult(
276
+ best_generator=best_generator,
277
+ history=history,
278
+ final_score=float(study.best_value)
279
+ if study.best_value is not None
280
+ else 0.0,
281
+ early_stopped=checker.get_state()["stopped"] if checker else False,
282
+ stop_reason=checker.get_state()["stop_reason"] if checker else None,
283
+ total_iterations=len(history),
284
+ total_evaluations=(
285
+ checker.get_state()["total_evaluations"]
286
+ if checker
287
+ else sum(len(h.individual_results) for h in history)
288
+ ),
289
+ )
290
+
291
+ def _score_prompt(
292
+ self,
293
+ prompt: str,
294
+ evaluator: Evaluator,
295
+ data_mapper: BasicDataMapper,
296
+ dataset: List[Dict[str, Any]],
297
+ ) -> Optional[IterationHistory]:
298
+ try:
299
+ eval_dataset = self._select_eval_subset(dataset)
300
+ temp_generator = LiteLLMGenerator(self.inference_model_name, prompt)
301
+
302
+ generated_outputs = [
303
+ temp_generator.generate(example, **self.inference_model_kwargs)
304
+ for example in eval_dataset
305
+ ]
306
+ eval_inputs = [
307
+ data_mapper.map(gen_out, ex)
308
+ for gen_out, ex in zip(generated_outputs, eval_dataset)
309
+ ]
310
+ results = evaluator.evaluate(eval_inputs)
311
+ avg_score = self.score_aggregator(results)
312
+ return IterationHistory(
313
+ prompt=prompt, average_score=avg_score, individual_results=results
314
+ )
315
+ except Exception as e:
316
+ logging.error(f"Failed to score prompt: {e}")
317
+ return None
318
+
319
+ def _infer_example_template(self, dataset: List[Dict[str, Any]]) -> str:
320
+ sample_size = min(self.template_infer_n_samples, max(1, len(dataset)))
321
+ sample = (
322
+ random.sample(dataset, sample_size)
323
+ if len(dataset) > sample_size
324
+ else dataset
325
+ )
326
+
327
+ # Build a minimal payload: include keys and a few short examples limited to those keys
328
+ keys: List[str] = sorted({k for ex in sample for k in ex.keys()})
329
+ trimmed_examples: List[Dict[str, Any]] = [
330
+ {k: str(ex.get(k, ""))[:500] for k in keys} for ex in sample
331
+ ]
332
+ user_payload = json.dumps(
333
+ {"keys": keys, "examples": trimmed_examples}, ensure_ascii=False
334
+ )
335
+
336
+ prompt_template = (
337
+ f"{self.teacher_system_prompt}\n\n"
338
+ "Available keys:\n{keys}\n\n"
339
+ "Examples (JSON):\n{examples_json}\n\n"
340
+ 'Respond ONLY with a JSON object like {{"example_template": "..."}}.'
341
+ )
342
+
343
+ teacher = LiteLLMGenerator(self.teacher_model_name, prompt_template)
344
+
345
+ last_err: Optional[Exception] = None
346
+ for attempt in range(self.teacher_infer_max_retries + 1):
347
+ try:
348
+ content = teacher.generate(
349
+ {"keys": ", ".join(keys), "examples_json": user_payload},
350
+ response_format={"type": "json_object"},
351
+ **self.teacher_model_kwargs,
352
+ )
353
+ template = self._parse_example_template_from_content(content)
354
+ if template:
355
+ return template
356
+ raise ValueError("Missing or empty 'example_template' in response")
357
+ except Exception as e:
358
+ last_err = e
359
+ if attempt < self.teacher_infer_max_retries:
360
+ time.sleep(self.teacher_infer_retry_sleep)
361
+ else:
362
+ break
363
+ raise RuntimeError(f"Teacher template inference failed: {last_err}")
364
+
365
+ @staticmethod
366
+ def _parse_example_template_from_content(content: str) -> Optional[str]:
367
+ # First try strict JSON
368
+ try:
369
+ data = json.loads(content)
370
+ tmpl = data.get("example_template")
371
+ if isinstance(tmpl, str) and tmpl.strip():
372
+ return tmpl
373
+ except Exception:
374
+ pass
375
+ # Try to extract JSON object containing example_template
376
+ try:
377
+ match = re.search(
378
+ r"\{[\s\S]*?\"example_template\"\s*:\s*\"[\s\S]*?\"[\s\S]*?\}", content
379
+ )
380
+ if match:
381
+ obj = json.loads(match.group(0))
382
+ tmpl = obj.get("example_template")
383
+ if isinstance(tmpl, str) and tmpl.strip():
384
+ return tmpl
385
+ except Exception:
386
+ pass
387
+ return None
388
+
389
+ def _select_eval_subset(
390
+ self, dataset: List[Dict[str, Any]]
391
+ ) -> List[Dict[str, Any]]:
392
+ if not self.eval_subset_size or self.eval_subset_size >= len(dataset):
393
+ return dataset
394
+ size = max(1, self.eval_subset_size)
395
+ if self.eval_subset_strategy == "first":
396
+ return dataset[:size]
397
+ elif self.eval_subset_strategy == "random":
398
+ return random.sample(dataset, size)
399
+ else:
400
+ return dataset
401
+
402
+ def _format_example(self, example: Dict[str, Any]) -> str:
403
+ if self.example_formatter:
404
+ return self.example_formatter(example)
405
+ template = self._runtime_example_template or self.example_template
406
+ if template:
407
+ try:
408
+ return template.format(**example)
409
+ except Exception:
410
+ pass
411
+ # Fallbacks when no template or failed formatting
412
+ if self.example_template_fields:
413
+ lines: List[str] = []
414
+ for key in self.example_template_fields:
415
+ if key in example:
416
+ label = self.field_aliases.get(key, key)
417
+ lines.append(f"{label}: {example[key]}")
418
+ if lines:
419
+ return "\n".join(lines)
420
+ # Final fallback: JSON dump of the example
421
+ return json.dumps(example, ensure_ascii=False)
422
+
423
+ def _build_few_shot_block(self, example_strings: List[str]) -> str:
424
+ block = self.example_separator.join(example_strings)
425
+ if self.few_shot_title:
426
+ return f"{self.few_shot_title}\n{block}"
427
+ return block
428
+
429
+ def _build_prompt(self, base_prompt: str, few_shot_block: str) -> str:
430
+ if self.prompt_builder:
431
+ return self.prompt_builder(base_prompt, [few_shot_block])
432
+ if not few_shot_block:
433
+ return base_prompt
434
+ # Escape braces in few-shot block to avoid str.format collisions
435
+ safe_block = self._escape_braces(few_shot_block)
436
+ if self.few_shot_position == "prepend":
437
+ return f"{safe_block}\n\n---\n\n{base_prompt}"
438
+ # default append
439
+ return f"{base_prompt}\n\n---\n\n{safe_block}\n\n---"
440
+
441
+ @staticmethod
442
+ def _escape_braces(text: str) -> str:
443
+ return text.replace("{", "{{").replace("}", "}}")
444
+
445
+ @staticmethod
446
+ def _default_score_aggregator(results: List[EvaluationResult]) -> float:
447
+ if not results:
448
+ return 0.0
449
+ return sum(r.score for r in results) / max(1, len(results))