agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,306 @@
1
+ """Child process a ``simulation-runner`` worker spawns per hosted job.
2
+
3
+ python -m fi.simulate.hosted.child_entrypoint <job.json> [--status-file PATH]
4
+
5
+ It runs the released SDK for the job's mode and submits results through
6
+ ``FutureAGIResultSink``. It is the only place hosted execution differs from a
7
+ local run — the simulation itself is the same ``SimulationRunner``/engine code.
8
+
9
+ Lifecycle is reported as newline-delimited JSON ``RunnerJobStatus`` objects, both
10
+ to stdout (the worker tails these for Temporal heartbeats) and to an optional
11
+ status file. SIGTERM triggers a graceful cancel + cleanup.
12
+
13
+ Slice 1 wires the chat mode only; the voice modes raise until their slices land.
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ import argparse
19
+ import asyncio
20
+ import json
21
+ import logging
22
+ import os
23
+ import signal
24
+ import sys
25
+ from datetime import datetime, timezone
26
+ from pathlib import Path
27
+ from typing import TYPE_CHECKING, Any
28
+
29
+ from fi.simulate.results.futureagi import FutureAGIResultSink
30
+ from fi.simulate.runtime.report import SimulationReport
31
+ from fi.simulate.runtime.run import RunStatus
32
+ from fi.simulate.runtime.runner import SimulationRunner
33
+
34
+ from .job import RunnerJobPhase, RunnerJobStatus, RunnerMode, StartRunnerJob
35
+ from .targets import resolve_chat_target
36
+
37
+ if TYPE_CHECKING:
38
+ from fi.simulate.runtime.spec import SimulationSpec
39
+
40
+ _HEARTBEAT_INTERVAL_SECONDS = 10.0
41
+ _CANCEL_GRACE_SECONDS = 30.0
42
+
43
+ logger = logging.getLogger("fi.simulate.hosted.runner")
44
+
45
+
46
+ def _job_log_fields(job: StartRunnerJob) -> dict[str, Any]:
47
+ fields: dict[str, Any] = {"job_id": job.job_id, "mode": job.mode.value}
48
+ if job.voice is not None:
49
+ target = dict(job.voice.agent_definition or {}).get("target") or {}
50
+ fields["provider"] = target.get("provider")
51
+ dataset = dict(job.voice.scenario or {}).get("dataset") or []
52
+ fields["cases"] = len(dataset)
53
+ if job.sink is not None:
54
+ fields["run_test_id"] = job.sink.run_test_id
55
+ fields["test_execution_id"] = job.sink.test_execution_id
56
+ return fields
57
+
58
+
59
+ class _StatusReporter:
60
+ def __init__(self, job_id: str, status_file: Path | None) -> None:
61
+ self._job_id = job_id
62
+ self._status_file = status_file
63
+
64
+ def emit(
65
+ self,
66
+ phase: RunnerJobPhase,
67
+ *,
68
+ detail: str | None = None,
69
+ report_hash: str | None = None,
70
+ submission_status: str | None = None,
71
+ ) -> None:
72
+ status = RunnerJobStatus(
73
+ job_id=self._job_id,
74
+ phase=phase,
75
+ detail=detail,
76
+ report_hash=report_hash,
77
+ submission_status=submission_status,
78
+ updated_at=datetime.now(timezone.utc),
79
+ )
80
+ line = status.model_dump_json()
81
+ print(line, flush=True)
82
+ if self._status_file is not None:
83
+ with self._status_file.open("a", encoding="utf-8") as handle:
84
+ handle.write(line + "\n")
85
+
86
+
87
+ def _load_job(path: Path) -> StartRunnerJob:
88
+ return StartRunnerJob.model_validate_json(path.read_text(encoding="utf-8"))
89
+
90
+
91
+ def _build_sink(job: StartRunnerJob) -> FutureAGIResultSink:
92
+ root = job.sink.root_directory or os.environ.get("FI_RUN_ROOT") or ".fagi/runs"
93
+ return FutureAGIResultSink(
94
+ root=root,
95
+ api_url=job.sink.api_url,
96
+ run_test_id=job.sink.run_test_id,
97
+ test_execution_id=job.sink.test_execution_id,
98
+ )
99
+
100
+
101
+ def _read_submission(run_directory: Path | None) -> dict[str, Any]:
102
+ if run_directory is None:
103
+ return {}
104
+ submission_path = run_directory / "submission.json"
105
+ if not submission_path.exists():
106
+ return {}
107
+ try:
108
+ return json.loads(submission_path.read_text(encoding="utf-8"))
109
+ except (ValueError, OSError):
110
+ return {}
111
+
112
+
113
+ def _build_voice_spec(job: StartRunnerJob) -> "SimulationSpec":
114
+ """Translate a voice job into a ``SimulationSpec`` so the voice run flows
115
+ through the same ``SimulationRunner`` spine as chat (plan §3). The typed voice
116
+ inputs ride in ``environment.config`` — secret-free, since providers are
117
+ referenced by ``*_env`` name, never raw values. ``transport.kind`` selects the
118
+ target adapter. The DID pool is leased by the runner activity (telephone
119
+ only), not here — the leased number arrives via the agent definition / params.
120
+ """
121
+ from fi.simulate.runtime import new_run_id
122
+ from fi.simulate.runtime.spec import (
123
+ AgentEndpointSpec,
124
+ EnvironmentSpec,
125
+ EvidencePolicy,
126
+ ExecutionPolicy,
127
+ SimulationSpec,
128
+ SimulatorPolicySpec,
129
+ TimeoutPolicy,
130
+ )
131
+ from fi.simulate.simulation.models import Scenario
132
+
133
+ cfg = job.voice
134
+ run_id = str((job.spec.run_id if job.spec else None) or new_run_id())
135
+ params = dict(cfg.params or {})
136
+ transport = (dict(cfg.agent_definition or {}).get("transport") or {})
137
+ transport_kind = transport.get("kind") or "livekit"
138
+
139
+ # The runner's outer deadline must clear the voice call's own budget.
140
+ run_seconds = max(
141
+ 300.0,
142
+ float(params.get("max_seconds", 45.0))
143
+ + float(params.get("connect_timeout", 15.0))
144
+ + float(params.get("readiness_timeout", 30.0))
145
+ + float(params.get("cleanup_timeout", 30.0))
146
+ + 60.0,
147
+ )
148
+
149
+ return SimulationSpec(
150
+ run_id=run_id,
151
+ environment=EnvironmentSpec(
152
+ adapter="voice",
153
+ world_kind="voice",
154
+ config={
155
+ "agent_definition": cfg.agent_definition,
156
+ "livekit_runtime": cfg.livekit_runtime,
157
+ "simulator": cfg.simulator,
158
+ "params": cfg.params,
159
+ },
160
+ ),
161
+ target=AgentEndpointSpec(adapter=transport_kind),
162
+ simulator=SimulatorPolicySpec(adapter="livekit_simulator"),
163
+ scenario=Scenario.model_validate(cfg.scenario),
164
+ execution=ExecutionPolicy(timeout=TimeoutPolicy(run_seconds=run_seconds)),
165
+ evidence=EvidencePolicy(),
166
+ )
167
+
168
+
169
+ async def _heartbeat(reporter: _StatusReporter) -> None:
170
+ while True:
171
+ await asyncio.sleep(_HEARTBEAT_INTERVAL_SECONDS)
172
+ reporter.emit(RunnerJobPhase.RUNNING, detail="heartbeat")
173
+
174
+
175
+ async def _execute(job: StartRunnerJob, reporter: _StatusReporter) -> int:
176
+ reporter.emit(RunnerJobPhase.PREPARING)
177
+ logger.info("hosted job start", extra=_job_log_fields(job))
178
+ sink = _build_sink(job)
179
+
180
+ if job.mode is RunnerMode.CHAT:
181
+ target = resolve_chat_target(job.spec)
182
+ run_coro = SimulationRunner().run(job.spec, target=target, result_sink=sink)
183
+ elif job.mode.is_voice:
184
+ run_coro = SimulationRunner().run(_build_voice_spec(job), result_sink=sink)
185
+ else:
186
+ raise NotImplementedError(f"runner mode not wired: {job.mode.value}")
187
+
188
+ run_task = asyncio.ensure_future(run_coro)
189
+ heartbeat_task = asyncio.ensure_future(_heartbeat(reporter))
190
+ reporter.emit(RunnerJobPhase.RUNNING)
191
+ try:
192
+ report: SimulationReport = await run_task
193
+ except asyncio.CancelledError:
194
+ reporter.emit(RunnerJobPhase.CANCELED, detail="cancelled")
195
+ logger.warning("hosted job cancelled", extra={"job_id": job.job_id})
196
+ # Cancelling this coroutine does not cancel ``run_task``; without an
197
+ # explicit cancel ``asyncio.run`` shutdown waits on it forever and the
198
+ # child leaks past SIGTERM.
199
+ run_task.cancel()
200
+ try:
201
+ await asyncio.wait({run_task}, timeout=_CANCEL_GRACE_SECONDS)
202
+ except asyncio.CancelledError:
203
+ pass
204
+ if not run_task.done():
205
+ os._exit(2)
206
+ raise
207
+ finally:
208
+ heartbeat_task.cancel()
209
+
210
+ reporter.emit(RunnerJobPhase.FINALIZING)
211
+ submission = _read_submission(sink.run_directory)
212
+ submission_status = submission.get("status")
213
+ run_completed = report.status is RunStatus.COMPLETED
214
+ # When a submission target is configured (a hosted run), a failed or omitted
215
+ # submission is a job failure — otherwise a broken upload reports as green.
216
+ submission_expected = bool(job.sink.run_test_id)
217
+ submission_ok = (not submission_expected) or submission_status == "submitted"
218
+ completed = run_completed and submission_ok
219
+ if completed:
220
+ detail = None
221
+ elif not run_completed:
222
+ detail = report.failure.code if report.failure else "run_failed"
223
+ else:
224
+ detail = f"submission_{submission_status or 'missing'}"
225
+ outcome_fields = {
226
+ **_job_log_fields(job),
227
+ "run_status": getattr(report.status, "value", str(report.status)),
228
+ "submission_status": submission_status,
229
+ "report_hash": report.report_hash,
230
+ "detail": detail,
231
+ }
232
+ if completed:
233
+ logger.info("hosted job completed", extra=outcome_fields)
234
+ else:
235
+ logger.error("hosted job failed", extra=outcome_fields)
236
+ reporter.emit(
237
+ RunnerJobPhase.COMPLETED if completed else RunnerJobPhase.FAILED,
238
+ detail=detail,
239
+ report_hash=report.report_hash,
240
+ submission_status=submission_status,
241
+ )
242
+ return 0 if completed else 1
243
+
244
+
245
+ def _install_cancellation(run_task_holder: dict[str, asyncio.Task[int]]) -> None:
246
+ loop = asyncio.get_running_loop()
247
+
248
+ def _cancel() -> None:
249
+ task = run_task_holder.get("task")
250
+ if task is not None and not task.done():
251
+ task.cancel()
252
+
253
+ for sig in (signal.SIGTERM, signal.SIGINT):
254
+ try:
255
+ loop.add_signal_handler(sig, _cancel)
256
+ except (NotImplementedError, ValueError):
257
+ pass
258
+
259
+
260
+ async def _main_async(job: StartRunnerJob, reporter: _StatusReporter) -> int:
261
+ holder: dict[str, asyncio.Task[int]] = {}
262
+ _install_cancellation(holder)
263
+ task = asyncio.ensure_future(_execute(job, reporter))
264
+ holder["task"] = task
265
+ try:
266
+ return await task
267
+ except asyncio.CancelledError:
268
+ return 2
269
+
270
+
271
+ def _configure_logging() -> None:
272
+ """The child runs with no logging config, so INFO seams (job start/outcome,
273
+ engine dispatch/join/stop_reason) were silently dropped by the WARNING-level
274
+ lastResort handler and never reached the runner's log capture."""
275
+ root = logging.getLogger()
276
+ if not root.handlers:
277
+ handler = logging.StreamHandler(sys.stderr)
278
+ handler.setFormatter(logging.Formatter("%(levelname)s:%(name)s:%(message)s"))
279
+ root.addHandler(handler)
280
+ root.setLevel(logging.WARNING)
281
+ logging.getLogger("fi.simulate").setLevel(logging.INFO)
282
+
283
+
284
+ def main(argv: list[str] | None = None) -> int:
285
+ _configure_logging()
286
+ parser = argparse.ArgumentParser(prog="fi.simulate.hosted.child_entrypoint")
287
+ parser.add_argument("job", help="path to the StartRunnerJob JSON file")
288
+ parser.add_argument("--status-file", default=None)
289
+ args = parser.parse_args(argv)
290
+
291
+ job = _load_job(Path(args.job))
292
+ status_file = Path(args.status_file) if args.status_file else None
293
+ reporter = _StatusReporter(job.job_id, status_file)
294
+
295
+ try:
296
+ return asyncio.run(_main_async(job, reporter))
297
+ except Exception as exc: # noqa: BLE001
298
+ logger.exception("hosted job crashed", extra={"job_id": job.job_id})
299
+ reporter.emit(
300
+ RunnerJobPhase.FAILED, detail=f"{type(exc).__name__}: {exc}"
301
+ )
302
+ return 1
303
+
304
+
305
+ if __name__ == "__main__":
306
+ sys.exit(main())
@@ -0,0 +1,150 @@
1
+ """Hosted-runner job contracts (plan §9.1).
2
+
3
+ A ``StartRunnerJob`` is the serializable unit the platform hands to a
4
+ ``simulation-runner`` worker. It embeds an immutable ``SimulationSpec`` plus the
5
+ result-sink target; it carries only ``SecretRef``s, never resolved secrets (the
6
+ runner resolves those into the child process environment). The child process
7
+ (``fi.simulate.hosted.child_entrypoint``) consumes exactly this model.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ from datetime import datetime
13
+ from enum import Enum
14
+ from typing import Protocol
15
+
16
+ from pydantic import BaseModel, Field, JsonValue, model_validator
17
+
18
+ from fi.simulate.runtime.spec import SecretRef, SimulationSpec
19
+
20
+ RUNNER_JOB_SCHEMA_VERSION = "futureagi.runner-job.v1"
21
+
22
+
23
+ class RunnerMode(str, Enum):
24
+ """Execution mode the child selects an engine for. Only the SIP mode
25
+ leases a phone-number slot; chat and WebRTC never touch the pool."""
26
+
27
+ CHAT = "chat"
28
+ VOICE_WEBRTC = "voice_webrtc"
29
+ VOICE_SIP = "voice_sip"
30
+
31
+ @property
32
+ def needs_phone(self) -> bool:
33
+ return self is RunnerMode.VOICE_SIP
34
+
35
+ @property
36
+ def is_voice(self) -> bool:
37
+ return self in {RunnerMode.VOICE_WEBRTC, RunnerMode.VOICE_SIP}
38
+
39
+
40
+ class ResultSinkConfig(BaseModel):
41
+ """Where the child submits results. ``test_execution_id`` is set for hosted
42
+ runs (the platform pre-creates the execution); leaving it unset preserves
43
+ the local create-then-submit behavior."""
44
+
45
+ api_url: str | None = None
46
+ run_test_id: str | None = None
47
+ test_execution_id: str | None = None
48
+ root_directory: str | None = None
49
+ secret_refs: dict[str, SecretRef] = Field(default_factory=dict)
50
+
51
+
52
+ class VoiceRunConfig(BaseModel):
53
+ """Voice runs use ``run_voice_simulation`` (LiveKit), not the chat
54
+ ``SimulationRunner``. This carries the typed inputs as JSON-round-trippable
55
+ dicts the child hydrates into ``AgentDefinition`` / ``LiveKitSimulatorRuntime``
56
+ / ``Scenario`` / ``SimulatorAgentDefinition``. ``transport.kind`` on the
57
+ agent definition selects webrtc vs sip."""
58
+
59
+ agent_definition: dict[str, JsonValue]
60
+ scenario: dict[str, JsonValue]
61
+ livekit_runtime: dict[str, JsonValue] | None = None
62
+ simulator: dict[str, JsonValue] | None = None
63
+ params: dict[str, JsonValue] = Field(default_factory=dict)
64
+
65
+
66
+ class StartRunnerJob(BaseModel):
67
+ schema_version: str = RUNNER_JOB_SCHEMA_VERSION
68
+ job_id: str
69
+ mode: RunnerMode = RunnerMode.CHAT
70
+ spec: SimulationSpec | None = None
71
+ voice: VoiceRunConfig | None = None
72
+ sink: ResultSinkConfig = Field(default_factory=ResultSinkConfig)
73
+ job_token_env: str | None = None
74
+ metadata: dict[str, JsonValue] = Field(default_factory=dict)
75
+
76
+ @model_validator(mode="after")
77
+ def _validate(self) -> "StartRunnerJob":
78
+ if self.schema_version != RUNNER_JOB_SCHEMA_VERSION:
79
+ raise ValueError(f"runner_job_version_unsupported: {self.schema_version}")
80
+ if self.mode is RunnerMode.CHAT and self.spec is None:
81
+ raise ValueError("chat runner job requires a spec")
82
+ if self.mode.is_voice and self.voice is None:
83
+ raise ValueError(f"{self.mode.value} runner job requires a voice config")
84
+ return self
85
+
86
+
87
+ class RunnerJobPhase(str, Enum):
88
+ PENDING = "pending"
89
+ PREPARING = "preparing"
90
+ RUNNING = "running"
91
+ FINALIZING = "finalizing"
92
+ COMPLETED = "completed"
93
+ FAILED = "failed"
94
+ CANCELED = "canceled"
95
+
96
+ @property
97
+ def terminal(self) -> bool:
98
+ return self in {
99
+ RunnerJobPhase.COMPLETED,
100
+ RunnerJobPhase.FAILED,
101
+ RunnerJobPhase.CANCELED,
102
+ }
103
+
104
+
105
+ class RunnerJobHandle(BaseModel):
106
+ job_id: str
107
+ run_id: str
108
+ pid: int | None = None
109
+ run_directory: str | None = None
110
+ metadata: dict[str, JsonValue] = Field(default_factory=dict)
111
+
112
+
113
+ class RunnerJobStatus(BaseModel):
114
+ job_id: str
115
+ phase: RunnerJobPhase
116
+ detail: str | None = None
117
+ report_hash: str | None = None
118
+ submission_status: str | None = None
119
+ updated_at: datetime
120
+
121
+
122
+ class RunnerReconcileResult(BaseModel):
123
+ reconciled: bool
124
+ orphan_ids: list[str] = Field(default_factory=list)
125
+ metadata: dict[str, JsonValue] = Field(default_factory=dict)
126
+
127
+
128
+ class HostedRunnerPort(Protocol):
129
+ """Scheduler-neutral port Temporal invokes (plan §9.1)."""
130
+
131
+ async def start(self, request: StartRunnerJob) -> RunnerJobHandle: ...
132
+
133
+ async def status(self, handle: RunnerJobHandle) -> RunnerJobStatus: ...
134
+
135
+ async def cancel(self, handle: RunnerJobHandle) -> None: ...
136
+
137
+ async def reconcile(self, handle: RunnerJobHandle) -> RunnerReconcileResult: ...
138
+
139
+
140
+ __all__ = [
141
+ "RUNNER_JOB_SCHEMA_VERSION",
142
+ "HostedRunnerPort",
143
+ "ResultSinkConfig",
144
+ "RunnerJobHandle",
145
+ "RunnerJobPhase",
146
+ "RunnerJobStatus",
147
+ "RunnerMode",
148
+ "RunnerReconcileResult",
149
+ "StartRunnerJob",
150
+ ]
@@ -0,0 +1,53 @@
1
+ """Resolve the agent-under-test target from a job's ``SimulationSpec.target``.
2
+
3
+ The runner never runs the target agent itself — the target is the customer's
4
+ deployed (or supplied) agent. For chat runs the target is a turn-based surface,
5
+ resolved through the one endpoint registry: ``spec.target.adapter`` names the
6
+ actor-source kind (``callable`` / ``python_callable`` / ``import_object`` /
7
+ ``factory`` / ``framework`` / ``system_prompt`` / ``http`` / …) and each
8
+ registered ``EndpointProfile`` carries the resolver. Adding a target kind is one
9
+ profile entry, no edits here (plan §4.1).
10
+
11
+ This is a HOSTED execution path (the runner runs it on our infra), so target
12
+ kinds that execute caller-supplied Python in-process (``callable`` /
13
+ ``python_callable`` / ``import_object`` / ``factory`` / ``framework``) are
14
+ **rejected here** — deny-by-default via ``EndpointProfile.runs_caller_code``.
15
+ Hosted runs must reach the agent as a deployed endpoint (``http`` / ``websocket``)
16
+ or through the sandboxed runtime. The only in-process escape is a trusted
17
+ operator-configured default target, opted in explicitly with
18
+ ``ALK_UNSAFE_INPROCESS_CODE_ACTORS`` — never set in prod for untrusted jobs.
19
+ """
20
+
21
+ from __future__ import annotations
22
+
23
+ from collections.abc import Callable
24
+ from typing import Any
25
+
26
+ from fi.simulate.agent.wrapper import AgentWrapper
27
+ from fi.simulate.runtime.spec import SimulationSpec
28
+
29
+
30
+ def resolve_chat_target(spec: SimulationSpec) -> Callable[..., Any] | AgentWrapper:
31
+ from fi.simulate.endpoints.actor_sources import (
32
+ ActorSourceError,
33
+ inprocess_code_allowed,
34
+ )
35
+ from fi.simulate.endpoints.profiles import get_profile
36
+
37
+ adapter = (spec.target.adapter or "").lower()
38
+ profile = get_profile(adapter)
39
+ if profile is None or not profile.is_turn_based_target:
40
+ raise ValueError(f"unsupported_chat_target_adapter: {spec.target.adapter}")
41
+ if profile.runs_caller_code and not inprocess_code_allowed():
42
+ raise ActorSourceError(
43
+ f"code_actor_denied_in_hosted: target {adapter!r} would run "
44
+ f"caller-supplied code in the runner process. Hosted runs must use a "
45
+ f"deployed endpoint (http/websocket) or the sandboxed runtime; "
46
+ f"in-process code is developer/local only."
47
+ )
48
+ return profile.resolve_target(
49
+ dict(spec.target.config or {}), spec.target.secret_refs, hosted=True
50
+ )
51
+
52
+
53
+ __all__ = ["resolve_chat_target"]
@@ -0,0 +1,5 @@
1
+ """Optional SDK instrumentation hooks (plan §6.6)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ __all__: list[str] = []
@@ -0,0 +1,122 @@
1
+ """FutureAGIObserver — LiveKit AgentSession event tap (plan §6.6 skeleton).
2
+
3
+ Attach an observer to a ``livekit.agents.AgentSession`` and it will
4
+ subscribe to a documented list of session events, translate each into
5
+ a ``CanonicalEvent`` from ``fi.simulate.runtime``, and hand it to a
6
+ pluggable sink (defaulting to an in-memory list so tests can assert
7
+ against emitted events). The real OTLP wiring lands with
8
+ ``OpenTelemetryEvidenceSource``; this observer only owns the SDK-side
9
+ event capture.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ import logging
15
+ from collections.abc import Callable
16
+ from typing import Any
17
+
18
+ from fi.simulate.runtime import CanonicalEvent, EventReliability
19
+
20
+ _SESSION_EVENT_MAP: dict[str, str] = {
21
+ "conversation_item_added": "transcript.final",
22
+ "user_state_changed": "speech.started",
23
+ "agent_state_changed": "session.ready",
24
+ "function_tool_execution_started": "tool.started",
25
+ "function_tool_execution_completed": "tool.completed",
26
+ "function_tool_execution_failed": "tool.failed",
27
+ "session_usage_updated": "usage.updated",
28
+ "close": "session.ended",
29
+ "error": "session.error",
30
+ }
31
+
32
+ logger = logging.getLogger(__name__)
33
+
34
+ EventSink = Callable[[CanonicalEvent], None]
35
+
36
+
37
+ class FutureAGIObserver:
38
+ def __init__(
39
+ self,
40
+ *,
41
+ run_id: str,
42
+ test_case_id: str,
43
+ sink: EventSink,
44
+ source: str = "livekit-observer",
45
+ ) -> None:
46
+ self._run_id = run_id
47
+ self._test_case_id = test_case_id
48
+ self._sink = sink
49
+ self._source = source
50
+ self._sequence = 0
51
+ self._attached = False
52
+
53
+ def attach(self, session: Any) -> "FutureAGIObserver":
54
+ if self._attached:
55
+ raise RuntimeError("observer_already_attached")
56
+ if not hasattr(session, "on"):
57
+ raise TypeError("session_incompatible: object has no on(event, callback)")
58
+ for session_event, canonical_type in _SESSION_EVENT_MAP.items():
59
+ handler = self._handler_for(session_event, canonical_type)
60
+ try:
61
+ session.on(session_event, handler)
62
+ except (AttributeError, ValueError):
63
+ logger.debug(
64
+ "livekit observer: session does not expose event",
65
+ extra={"event": session_event},
66
+ )
67
+ self._attached = True
68
+ return self
69
+
70
+ def emit(
71
+ self,
72
+ event_type: str,
73
+ payload: dict[str, Any] | None = None,
74
+ *,
75
+ reliability: EventReliability = EventReliability.RELIABLE,
76
+ ) -> CanonicalEvent:
77
+ self._sequence += 1
78
+ event = CanonicalEvent.create(
79
+ run_id=self._run_id,
80
+ test_case_id=self._test_case_id,
81
+ event_type=event_type,
82
+ source=self._source,
83
+ sequence=self._sequence,
84
+ reliability=reliability,
85
+ payload=payload or {},
86
+ )
87
+ self._sink(event)
88
+ return event
89
+
90
+ def _handler_for(self, session_event: str, canonical_type: str) -> Callable[..., None]:
91
+ def handler(*args: Any, **kwargs: Any) -> None:
92
+ payload = _summarize_payload(session_event, args, kwargs)
93
+ self.emit(canonical_type, payload)
94
+
95
+ return handler
96
+
97
+
98
+ def _summarize_payload(
99
+ session_event: str,
100
+ args: tuple[Any, ...],
101
+ kwargs: dict[str, Any],
102
+ ) -> dict[str, Any]:
103
+ def _describe(value: Any) -> Any:
104
+ if hasattr(value, "model_dump"):
105
+ try:
106
+ return value.model_dump(mode="json", exclude_none=True)
107
+ except Exception: # noqa: BLE001
108
+ pass
109
+ if hasattr(value, "__dict__"):
110
+ return {"repr": type(value).__name__}
111
+ if isinstance(value, (str, int, float, bool)) or value is None:
112
+ return value
113
+ return type(value).__name__
114
+
115
+ return {
116
+ "session_event": session_event,
117
+ "args": [_describe(item) for item in args],
118
+ "kwargs": {key: _describe(value) for key, value in kwargs.items()},
119
+ }
120
+
121
+
122
+ __all__ = ["FutureAGIObserver"]