agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,322 @@
1
+ import logging
2
+ import time
3
+ from typing import Any, Dict, List, Optional
4
+
5
+ # Import GEPA's core components. A try/except block makes this a soft dependency.
6
+ try:
7
+ import gepa
8
+ from gepa.core.adapter import GEPAAdapter, EvaluationBatch, DataInst
9
+ except ImportError:
10
+ raise ImportError(
11
+ "To use GEPAOptimizer, please install the 'gepa' library with: pip install gepa"
12
+ )
13
+
14
+ from ..base.base_optimizer import BaseOptimizer
15
+ from ..datamappers.basic_mapper import BasicDataMapper
16
+ from ..base.evaluator import Evaluator
17
+ from ..generators.litellm import LiteLLMGenerator
18
+ from ..types import OptimizationResult, IterationHistory
19
+ from ..utils.early_stopping import (
20
+ EarlyStoppingConfig,
21
+ EarlyStoppingChecker,
22
+ EarlyStoppingException,
23
+ )
24
+
25
+ logger = logging.getLogger(__name__)
26
+
27
+
28
+ class _InternalGEPAAdapter(GEPAAdapter[DataInst, Dict[str, Any], Dict[str, Any]]):
29
+ """
30
+ An internal adapter that translates our framework's components (Evaluator,
31
+ DataMapper) into the interface GEPA's optimization engine expects.
32
+ """
33
+
34
+ def __init__(
35
+ self,
36
+ generator_model: str,
37
+ evaluator: Evaluator,
38
+ data_mapper: BasicDataMapper,
39
+ history_list: List[IterationHistory],
40
+ early_stopping_checker: Optional[EarlyStoppingChecker] = None,
41
+ ):
42
+ self.generator_model = generator_model
43
+ self.evaluator = evaluator
44
+ self.data_mapper = data_mapper
45
+ self.history_list = history_list
46
+ self.early_stopping_checker = early_stopping_checker
47
+ logger.info(f"Initialized with generator_model: {generator_model}")
48
+
49
+ def evaluate(
50
+ self,
51
+ batch: List[Dict[str, Any]],
52
+ candidate: Dict[str, str],
53
+ capture_traces: bool = False,
54
+ ) -> EvaluationBatch[Dict[str, Any], Dict[str, Any]]:
55
+ """
56
+ This method is called by GEPA during its optimization loop. It uses
57
+ our framework's components to perform the evaluation.
58
+ """
59
+ eval_start_time = time.time()
60
+ logger.info("Starting evaluation for a candidate prompt.")
61
+
62
+ # GEPA provides the prompt as the first (and only) value in the candidate dict.
63
+ prompt_text = next(iter(candidate.values()))
64
+ logger.info(f"Evaluating prompt: '{prompt_text[:100]}...'")
65
+ logger.info(f"Batch size: {len(batch)}")
66
+
67
+ temp_generator = LiteLLMGenerator(
68
+ model=self.generator_model, prompt_template=prompt_text
69
+ )
70
+
71
+ logger.info("Generating outputs...")
72
+ gen_start_time = time.time()
73
+ generated_outputs = [temp_generator.generate(example) for example in batch]
74
+ gen_end_time = time.time()
75
+ logger.info(
76
+ f"Output generation finished in {gen_end_time - gen_start_time:.2f}s."
77
+ )
78
+
79
+ logger.info("Mapping evaluation inputs...")
80
+ eval_inputs = [
81
+ self.data_mapper.map(gen_out, ex)
82
+ for gen_out, ex in zip(generated_outputs, batch)
83
+ ]
84
+
85
+ logger.info("Evaluating generated outputs...")
86
+ evaluator_start_time = time.time()
87
+ results = self.evaluator.evaluate(eval_inputs)
88
+ evaluator_end_time = time.time()
89
+ logger.info(
90
+ f"Evaluation with framework evaluator finished in {evaluator_end_time - evaluator_start_time:.2f}s."
91
+ )
92
+
93
+ scores = [res.score for res in results]
94
+ logger.info(f"Scores: {scores}")
95
+ outputs = [
96
+ {"generated_text": out, "full_result": res.model_dump()}
97
+ for out, res in zip(generated_outputs, results)
98
+ ]
99
+
100
+ # capture iteration history
101
+ avg_score = sum(scores) / len(scores) if scores else 0.0
102
+ self.history_list.append(
103
+ IterationHistory(
104
+ prompt=prompt_text,
105
+ average_score=avg_score,
106
+ individual_results=results,
107
+ )
108
+ )
109
+
110
+ # Check early stopping
111
+ if self.early_stopping_checker:
112
+ if self.early_stopping_checker.should_stop(avg_score, len(batch)):
113
+ reason = self.early_stopping_checker.get_state()["stop_reason"]
114
+ logger.info(f"Early stopping triggered: {reason}")
115
+ raise EarlyStoppingException(reason)
116
+
117
+ trajectories = []
118
+ if capture_traces:
119
+ logger.info("Capturing traces.")
120
+ for i in range(len(batch)):
121
+ trajectories.append(
122
+ {
123
+ "inputs": batch[i],
124
+ "generated_output": generated_outputs[i],
125
+ "evaluation_result": results[i].model_dump(),
126
+ }
127
+ )
128
+
129
+ eval_end_time = time.time()
130
+ logger.info(f"Evaluation finished in {eval_end_time - eval_start_time:.2f}s.")
131
+ return EvaluationBatch(
132
+ outputs=outputs, scores=scores, trajectories=trajectories
133
+ )
134
+
135
+ def make_reflective_dataset(
136
+ self,
137
+ candidate: Dict[str, str],
138
+ eval_batch: EvaluationBatch,
139
+ components_to_update: List[str],
140
+ ) -> Dict[str, List[Dict[str, Any]]]:
141
+ """
142
+ Creates the dataset for GEPA's reflective LLM to analyze.
143
+ """
144
+ logger.info("Creating reflective dataset.")
145
+ reflective_data = {comp: [] for comp in components_to_update}
146
+
147
+ if not eval_batch.trajectories:
148
+ logger.warning("No trajectories found to create reflective dataset.")
149
+ return reflective_data
150
+
151
+ logger.info(f"Processing {len(eval_batch.trajectories)} trajectories.")
152
+ for trajectory in eval_batch.trajectories:
153
+ result = trajectory.get("evaluation_result", {})
154
+ score = result.get("score", 0.0)
155
+ reason = result.get("reason", "No reason provided.")
156
+
157
+ if score >= 0.8:
158
+ feedback = f"This output was successful (score={score:.2f}). The reasoning for this score was: {reason}"
159
+ else:
160
+ feedback = f"This output performed poorly (score={score:.2f}). The key reason for the failure was: {reason}. The prompt needs to be improved to avoid this specific failure mode."
161
+
162
+ example = {
163
+ "Inputs": trajectory.get("inputs", {}),
164
+ "Generated Outputs": trajectory.get("generated_output", ""),
165
+ "Feedback": feedback,
166
+ }
167
+
168
+ for comp in components_to_update:
169
+ reflective_data[comp].append(example)
170
+
171
+ logger.info(
172
+ f"Reflective dataset created for components: {components_to_update}"
173
+ )
174
+ return reflective_data
175
+
176
+
177
+ class GEPAOptimizer(BaseOptimizer):
178
+ """
179
+ An adapter that integrates the powerful GEPA evolutionary optimization
180
+ algorithm into the prompt-optimizer framework.
181
+ """
182
+
183
+ def __init__(self, reflection_model: str, generator_model: str = "gpt-4o-mini"):
184
+ """
185
+ Initializes the GEPA Optimizer wrapper.
186
+
187
+ Args:
188
+ reflection_model (str): The name of a powerful LLM (e.g., "gpt-4-turbo")
189
+ that GEPA will use for its reflection and mutation steps.
190
+ generator_model (str): The name of the model that will be used by the
191
+ prompts being optimized (the "task language model").
192
+ """
193
+ self.reflection_model = reflection_model
194
+ self.generator_model = generator_model
195
+ logger.info(
196
+ f"Initialized with reflection_model: {reflection_model}, generator_model: {generator_model}"
197
+ )
198
+
199
+ def optimize(
200
+ self,
201
+ evaluator: Evaluator,
202
+ data_mapper: BasicDataMapper,
203
+ dataset: List[Dict[str, Any]],
204
+ initial_prompts: List[str],
205
+ max_metric_calls: Optional[int] = 150,
206
+ early_stopping: Optional[EarlyStoppingConfig] = None,
207
+ ) -> OptimizationResult:
208
+ opt_start_time = time.time()
209
+ logger.info("--- Starting GEPA Prompt Optimization ---")
210
+ logger.info(f"Dataset size: {len(dataset)}")
211
+ logger.info(f"Initial prompts: {initial_prompts}")
212
+ logger.info(f"Max metric calls: {max_metric_calls}")
213
+
214
+ # Initialize early stopping checker
215
+ checker = None
216
+ if early_stopping and early_stopping.is_enabled():
217
+ checker = EarlyStoppingChecker(early_stopping)
218
+ logger.info(f"Early stopping enabled: {early_stopping}")
219
+
220
+ if not initial_prompts:
221
+ raise ValueError("Initial prompts list cannot be empty for GEPAOptimizer.")
222
+ history: List[IterationHistory] = []
223
+ # 1. Create the internal adapter that bridges our framework to GEPA
224
+ logger.info("Creating internal GEPA adapter...")
225
+ adapter = _InternalGEPAAdapter(
226
+ generator_model=self.generator_model,
227
+ evaluator=evaluator,
228
+ data_mapper=data_mapper,
229
+ history_list=history,
230
+ early_stopping_checker=checker,
231
+ )
232
+
233
+ # 2. Prepare the inputs for gepa.optimize
234
+ seed_candidate = {"prompt": initial_prompts[0]}
235
+ logger.info(f"Seed candidate for GEPA: {seed_candidate}")
236
+
237
+ # 3. Call the external GEPA library's optimize function
238
+ logger.info("Calling gepa.optimize...")
239
+ gepa_start_time = time.time()
240
+
241
+ try:
242
+ gepa_result = gepa.optimize(
243
+ seed_candidate=seed_candidate,
244
+ trainset=dataset,
245
+ valset=dataset,
246
+ adapter=adapter,
247
+ reflection_lm=self.reflection_model,
248
+ max_metric_calls=max_metric_calls,
249
+ display_progress_bar=True,
250
+ )
251
+ gepa_end_time = time.time()
252
+ logger.info(
253
+ f"gepa.optimize finished in {gepa_end_time - gepa_start_time:.2f}s."
254
+ )
255
+ logger.info(
256
+ f"GEPA result best score: {gepa_result.val_aggregate_scores[gepa_result.best_idx]}"
257
+ )
258
+ logger.info(f"GEPA best candidate: {gepa_result.best_candidate}")
259
+
260
+ logger.info(f"Captured {len(history)} iterations in history.")
261
+ # 4. Translate GEPA's result back into our framework's standard format
262
+ logger.info("Translating GEPA result to OptimizationResult...")
263
+
264
+ final_best_generator = LiteLLMGenerator(
265
+ model=self.generator_model,
266
+ prompt_template=gepa_result.best_candidate.get("prompt", ""),
267
+ )
268
+
269
+ # Build result with early stopping metadata
270
+ result = OptimizationResult(
271
+ best_generator=final_best_generator,
272
+ history=history,
273
+ final_score=gepa_result.val_aggregate_scores[gepa_result.best_idx],
274
+ early_stopped=False,
275
+ stop_reason=None,
276
+ total_iterations=len(history),
277
+ total_evaluations=(
278
+ checker.get_state()["total_evaluations"]
279
+ if checker
280
+ else sum(len(h.individual_results) for h in history)
281
+ ),
282
+ )
283
+
284
+ except EarlyStoppingException as e:
285
+ gepa_end_time = time.time()
286
+ logger.info(
287
+ f"GEPA stopped early after {gepa_end_time - gepa_start_time:.2f}s: {e.reason}"
288
+ )
289
+
290
+ # Use best from history
291
+ if not history:
292
+ raise RuntimeError(
293
+ "Early stopping triggered before any evaluations completed"
294
+ )
295
+
296
+ best_history = max(history, key=lambda h: h.average_score)
297
+ final_best_generator = LiteLLMGenerator(
298
+ model=self.generator_model,
299
+ prompt_template=best_history.prompt,
300
+ )
301
+
302
+ result = OptimizationResult(
303
+ best_generator=final_best_generator,
304
+ history=history,
305
+ final_score=best_history.average_score,
306
+ early_stopped=True,
307
+ stop_reason=e.reason,
308
+ total_iterations=len(history),
309
+ total_evaluations=(
310
+ checker.get_state()["total_evaluations"]
311
+ if checker
312
+ else sum(len(h.individual_results) for h in history)
313
+ ),
314
+ )
315
+
316
+ opt_end_time = time.time()
317
+ logger.info(
318
+ f"--- GEPA Prompt Optimization finished in {opt_end_time - opt_start_time:.2f}s ---"
319
+ )
320
+ logger.info(f"Final best score: {result.final_score}")
321
+
322
+ return result
@@ -0,0 +1,243 @@
1
+ import json
2
+ import random
3
+ from typing import Any, Dict, List, Optional
4
+
5
+ from pydantic import BaseModel, Field, ValidationError
6
+
7
+ from ..base.base_optimizer import BaseOptimizer
8
+ from ..datamappers.basic_mapper import BasicDataMapper
9
+ from ..base.evaluator import Evaluator
10
+ from ..generators.litellm import LiteLLMGenerator
11
+ from ..types import IterationHistory, OptimizationResult
12
+ from ..utils.early_stopping import EarlyStoppingConfig, EarlyStoppingChecker
13
+ import logging
14
+
15
+ logger = logging.getLogger(__name__)
16
+ # ==============================================================================
17
+ # Prompts and Pydantic Models for the Teacher LLM (Meta-Model)
18
+ # ==============================================================================
19
+
20
+
21
+ META_PROMPT_TEMPLATE = """
22
+ You are a world-class expert in prompt engineering. Your task is to diagnose and optimize a given prompt based on its performance on a set of test cases.
23
+
24
+ ### Current Prompt
25
+ The following is the current prompt being evaluated:
26
+ ---
27
+ {current_prompt}
28
+ ---
29
+
30
+ ### Previous Failed Attempts
31
+ You have already tried the following prompts, but they performed worse than the current one. Analyze why they failed to avoid repeating mistakes.
32
+ ---
33
+ {other_attempts}
34
+ ---
35
+
36
+ ### Performance Data
37
+ The current prompt was run on a set of examples, and here are the results. Pay close attention to the examples with low scores.
38
+ ---
39
+ {annotated_results}
40
+ ---
41
+
42
+ ### Task Description
43
+ {task_description}
44
+
45
+ ### Your Task
46
+ Think step-by-step to generate an improved prompt:
47
+ 1. **Analyze Failures:** Deeply analyze the failing examples. What patterns do you see? Is the prompt too vague, too restrictive, or missing key instructions?
48
+ 2. **Formulate a Hypothesis:** Based on your analysis, state a clear hypothesis for how to improve the prompt. For example, "My hypothesis is that adding a chain-of-thought instruction will improve reasoning on multi-step problems."
49
+ 3. **Generate Improved Prompt:** Rewrite the *entire* prompt, implementing your hypothesis. The new prompt should be a complete replacement for the current one.
50
+
51
+ Return ONLY a valid JSON object with two keys: "hypothesis" (your string hypothesis) and "improved_prompt" (the complete new prompt string).
52
+ """
53
+
54
+
55
+ class MetaPromptOutput(BaseModel):
56
+ hypothesis: str = Field(
57
+ description="The hypothesis for why the new prompt will be better."
58
+ )
59
+ improved_prompt: str = Field(description="The complete, new, improved prompt.")
60
+
61
+
62
+ # ==============================================================================
63
+ # The MetaPrompt Optimizer Class
64
+ # ==============================================================================
65
+
66
+
67
+ class MetaPromptOptimizer(BaseOptimizer):
68
+ """
69
+ Optimizes a prompt by using a powerful "teacher" LLM to analyze its
70
+ performance and rewrite it. This is inspired by the `promptim` library.
71
+ """
72
+
73
+ def __init__(
74
+ self,
75
+ teacher_generator: LiteLLMGenerator,
76
+ task_model: Optional[str] = None,
77
+ ):
78
+ """
79
+ Initializes the MetaPrompt Optimizer.
80
+
81
+ Args:
82
+ teacher_generator: A powerful generator (e.g., GPT-4o, Claude 3 Opus)
83
+ used to analyze performance and generate new prompts.
84
+ task_model: Model used to run candidate prompts while scoring them.
85
+ Defaults to the teacher generator's model.
86
+ """
87
+ self.teacher = teacher_generator
88
+ self.task_model = task_model
89
+
90
+ def optimize(
91
+ self,
92
+ evaluator: Evaluator,
93
+ data_mapper: BasicDataMapper,
94
+ dataset: List[Dict[str, Any]],
95
+ initial_prompts: List[str],
96
+ task_description: str = "I want to improve my prompt.",
97
+ num_rounds: Optional[int] = 5,
98
+ eval_subset_size: Optional[int] = 40,
99
+ early_stopping: Optional[EarlyStoppingConfig] = None,
100
+ ) -> OptimizationResult:
101
+ logger.info("--- Starting Meta-Prompt Optimization ---")
102
+
103
+ # Initialize early stopping checker
104
+ checker = None
105
+ if early_stopping and early_stopping.is_enabled():
106
+ checker = EarlyStoppingChecker(early_stopping)
107
+ logger.info(f"Early stopping enabled: {early_stopping}")
108
+
109
+ if not initial_prompts:
110
+ raise ValueError("Initial prompts list cannot be empty.")
111
+
112
+ current_prompt = initial_prompts[0]
113
+ best_prompt = current_prompt
114
+ best_score = -1.0
115
+ history: List[IterationHistory] = []
116
+ previous_attempts = set()
117
+
118
+ for round_num in range(num_rounds):
119
+ logger.info(
120
+ f"\n--- Starting Optimization Round {round_num + 1}/{num_rounds} ---"
121
+ )
122
+ logger.info(f"Current best prompt:\n{current_prompt}")
123
+
124
+ # 1. Evaluate the current prompt on a subset of data
125
+ eval_subset = random.sample(dataset, min(len(dataset), eval_subset_size))
126
+ iteration_history = self._score_prompt(
127
+ current_prompt, evaluator, data_mapper, eval_subset
128
+ )
129
+
130
+ if not iteration_history:
131
+ logger.warning("Evaluation of current prompt failed. Skipping round.")
132
+ continue
133
+
134
+ history.append(iteration_history)
135
+ current_score = iteration_history.average_score
136
+
137
+ if current_score > best_score:
138
+ best_score = current_score
139
+ best_prompt = current_prompt
140
+ logger.info(f"New best score found: {best_score:.4f}")
141
+
142
+ # Check early stopping
143
+ if checker:
144
+ num_evals = len(eval_subset)
145
+ if checker.should_stop(current_score, num_evals):
146
+ logger.info(
147
+ f"Early stopping triggered: {checker.get_state()['stop_reason']}"
148
+ )
149
+ break
150
+
151
+ # 2. Use the teacher model to generate a new, improved prompt
152
+ annotated_results_str = self._format_results(iteration_history, eval_subset)
153
+
154
+ # Format previous attempts for the meta-prompt
155
+ other_attempts_str = (
156
+ "\n---\n".join(list(previous_attempts)) if previous_attempts else "N/A"
157
+ )
158
+
159
+ meta_prompt = META_PROMPT_TEMPLATE.format(
160
+ current_prompt=current_prompt,
161
+ other_attempts=other_attempts_str,
162
+ annotated_results=annotated_results_str,
163
+ task_description=task_description,
164
+ )
165
+
166
+ logger.debug("Generating new prompt with meta-prompt...")
167
+ new_prompt_json = self.teacher.generate(
168
+ prompt_vars={"prompt": meta_prompt},
169
+ response_format={"type": "json_object"},
170
+ )
171
+
172
+ try:
173
+ parsed_output = MetaPromptOutput.model_validate_json(new_prompt_json)
174
+ logger.info(f"Teacher's Hypothesis: {parsed_output.hypothesis}")
175
+ previous_attempts.add(current_prompt)
176
+ current_prompt = parsed_output.improved_prompt
177
+ except (ValidationError, json.JSONDecodeError) as e:
178
+ logger.error(
179
+ f"Failed to parse new prompt from teacher model, keeping current prompt. Error: {e}"
180
+ )
181
+
182
+ final_best_generator = LiteLLMGenerator(self.teacher.model_name, best_prompt)
183
+
184
+ # Build result with early stopping metadata
185
+ return OptimizationResult(
186
+ best_generator=final_best_generator,
187
+ history=history,
188
+ final_score=best_score,
189
+ early_stopped=checker.get_state()["stopped"] if checker else False,
190
+ stop_reason=checker.get_state()["stop_reason"] if checker else None,
191
+ total_iterations=len(history),
192
+ total_evaluations=(
193
+ checker.get_state()["total_evaluations"]
194
+ if checker
195
+ else sum(len(h.individual_results) for h in history)
196
+ ),
197
+ )
198
+
199
+ def _score_prompt(
200
+ self,
201
+ prompt: str,
202
+ evaluator: Evaluator,
203
+ data_mapper: BasicDataMapper,
204
+ dataset: List[Dict[str, Any]],
205
+ ) -> IterationHistory | None:
206
+ """Scores a single prompt and returns its history."""
207
+ try:
208
+ temp_generator = LiteLLMGenerator(
209
+ self.task_model or self.teacher.model_name, prompt
210
+ )
211
+ generated_outputs = [
212
+ temp_generator.generate(example) for example in dataset
213
+ ]
214
+ eval_inputs = [
215
+ data_mapper.map(gen_out, ex)
216
+ for gen_out, ex in zip(generated_outputs, dataset)
217
+ ]
218
+ results = evaluator.evaluate(eval_inputs)
219
+ avg_score = (
220
+ sum(res.score for res in results) / len(results) if results else 0.0
221
+ )
222
+ return IterationHistory(
223
+ prompt=prompt, average_score=avg_score, individual_results=results
224
+ )
225
+ except Exception as e:
226
+ logger.error(f"Failed to score prompt: {e}")
227
+ return None
228
+
229
+ def _format_results(
230
+ self, iteration_history: IterationHistory, dataset: List[Dict[str, Any]]
231
+ ) -> str:
232
+ """Formats the evaluation results into a string for the meta-prompt."""
233
+ formatted_lines = []
234
+ for i, result in enumerate(iteration_history.individual_results):
235
+ example_input = dataset[i]
236
+ formatted_lines.append(f"Example {i + 1}:")
237
+ formatted_lines.append(
238
+ f" Input: {json.dumps(example_input, ensure_ascii=False)}"
239
+ )
240
+ formatted_lines.append(f" Score: {result.score:.2f}")
241
+ formatted_lines.append(f" Reason: {result.reason}")
242
+ formatted_lines.append("---")
243
+ return "\n".join(formatted_lines)