agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,253 @@
1
+ """Where an agent comes from, and how a session reaches it.
2
+
3
+ A folder of source code is one kind of agent, not the only kind. The same agent may arrive as a
4
+ provider connection with a system prompt and a tool schema, as a platform definition, or as a
5
+ spec somebody pasted in. The stage that reads an agent is the same in all of those cases; what
6
+ differs is where it looks and what it is allowed to touch.
7
+
8
+ So the method stays in the skill and the location lives here. Supporting a new kind of agent is
9
+ registering one class, not editing any stage.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ import json
15
+ import subprocess
16
+ from collections.abc import Callable
17
+ from dataclasses import dataclass, field
18
+ from pathlib import Path
19
+ from typing import Any, Protocol
20
+
21
+
22
+ class AgentSource(Protocol):
23
+ """Everything a stage needs in order to reach one agent."""
24
+
25
+ kind: str
26
+ name: str
27
+
28
+ def workdir(self) -> Path:
29
+ """The directory the session runs in."""
30
+
31
+ def builtin_tools(self) -> tuple[str, ...]:
32
+ """Built-in tools this source needs granted."""
33
+
34
+ def servers(self) -> dict[str, Any]:
35
+ """In-process tool servers this source provides, if any."""
36
+
37
+ def briefing(self) -> str:
38
+ """What to tell the model about where this agent's truth lives."""
39
+
40
+
41
+ @dataclass
42
+ class RepoSource:
43
+ """An agent that exists as source code on disk."""
44
+
45
+ name: str
46
+ root: Path
47
+ kind: str = "repo"
48
+
49
+ def workdir(self) -> Path:
50
+ return self.root
51
+
52
+ def builtin_tools(self) -> tuple[str, ...]:
53
+ return ("Read", "Glob", "Grep")
54
+
55
+ def servers(self) -> dict[str, Any]:
56
+ return {}
57
+
58
+ def briefing(self) -> str:
59
+ ignored = {
60
+ ".git",
61
+ ".venv",
62
+ "node_modules",
63
+ "__pycache__",
64
+ "artifacts",
65
+ "build",
66
+ "dist",
67
+ }
68
+ indexed: list[str] = []
69
+ for path in sorted(self.root.rglob("*")):
70
+ try:
71
+ relative = path.relative_to(self.root)
72
+ except ValueError:
73
+ continue
74
+ if any(part in ignored for part in relative.parts) or not path.is_file():
75
+ continue
76
+ indexed.append(relative.as_posix())
77
+ if len(indexed) >= 240:
78
+ break
79
+ return (
80
+ f"This agent is a repository at {self.root}. Its truth is the source code: the tool "
81
+ "registrations, the function signatures, the validation logic, and whatever holds "
82
+ "its data. Read it with Read, Glob and Grep. Documentation describes intent; the "
83
+ "code describes behaviour, and where they disagree the code wins. Never glob the "
84
+ "entire repository: dependency caches such as .venv and node_modules are irrelevant. "
85
+ "Start from this pre-indexed source/config file list and read only relevant files:\n"
86
+ + "\n".join(f"- {name}" for name in indexed)
87
+ )
88
+
89
+
90
+ @dataclass
91
+ class GitHubSource(RepoSource):
92
+ """A public GitHub repository cloned into this harness session."""
93
+
94
+ url: str = ""
95
+ kind: str = "github"
96
+
97
+ def briefing(self) -> str:
98
+ return (
99
+ f"This agent was cloned from {self.url or 'GitHub'} into {self.root}. Its truth is "
100
+ "the cloned source code: the tool registrations, function signatures, validation "
101
+ "logic, and whatever holds its data. Read it with Read, Glob and Grep."
102
+ )
103
+
104
+
105
+ def clone_github_repository(url: str, destination: Path) -> Path:
106
+ """Shallow-clone one public GitHub repository into a session-owned directory."""
107
+ from .github import parse_github_location
108
+
109
+ try:
110
+ location = parse_github_location(url)
111
+ except ValueError as exc:
112
+ raise ValueError(
113
+ "use a public HTTPS GitHub repository or branch URL, such as "
114
+ "https://github.com/owner/repo/tree/branch"
115
+ ) from exc
116
+ if destination.exists():
117
+ raise ValueError(f"the session source directory already exists: {destination}")
118
+
119
+ destination.parent.mkdir(parents=True, exist_ok=True)
120
+ command = ["git", "clone", "--depth", "1"]
121
+ if location.ref:
122
+ command.extend(["--branch", location.ref])
123
+ command.extend([location.clone_url, str(destination)])
124
+ completed = subprocess.run(
125
+ command,
126
+ capture_output=True,
127
+ check=False,
128
+ text=True,
129
+ )
130
+ if completed.returncode:
131
+ detail = completed.stderr.strip() or "git clone failed"
132
+ raise RuntimeError(detail)
133
+ return destination
134
+
135
+
136
+ @dataclass
137
+ class SpecSource:
138
+ """An agent supplied directly as a prompt and a tool schema, with no repository.
139
+
140
+ This is the shape a hosted provider gives back, so it is also the fallback whenever a
141
+ connection can be read once and handed over as text.
142
+ """
143
+
144
+ name: str
145
+ system_prompt: str
146
+ tool_schema: list[dict[str, Any]] = field(default_factory=list)
147
+ data: dict[str, Any] = field(default_factory=dict)
148
+ scratch: Path = Path(".")
149
+ kind: str = "spec"
150
+
151
+ def workdir(self) -> Path:
152
+ return self.scratch
153
+
154
+ def builtin_tools(self) -> tuple[str, ...]:
155
+ return ()
156
+
157
+ def servers(self) -> dict[str, Any]:
158
+ return {}
159
+
160
+ def briefing(self) -> str:
161
+ parts = [
162
+ "This agent is supplied as a definition, not a repository. Everything knowable "
163
+ "about it is below; there is no code to open, so do not guess at anything absent.",
164
+ f"SYSTEM PROMPT:\n{self.system_prompt}",
165
+ ]
166
+ if self.tool_schema:
167
+ parts.append(
168
+ f"TOOL SCHEMA:\n{json.dumps(self.tool_schema, indent=2)[:6000]}"
169
+ )
170
+ if self.data:
171
+ parts.append(f"DATA:\n{json.dumps(self.data, indent=2)[:6000]}")
172
+ return "\n\n".join(parts)
173
+
174
+
175
+ @dataclass
176
+ class ProviderSource:
177
+ """A sanitized definition fetched from an externally hosted provider.
178
+
179
+ A connect-only provider agent has no repository in the sandbox. Representing its empty
180
+ source directory as a :class:`RepoSource` gives an authoring model filesystem tools and can
181
+ make it wander outside that directory looking for an implementation. The provider profile
182
+ is the complete source of truth for this mode, so expose only that profile and no file tools.
183
+ """
184
+
185
+ name: str
186
+ profile: dict[str, Any]
187
+ scratch: Path = Path(".")
188
+ kind: str = "provider"
189
+
190
+ def workdir(self) -> Path:
191
+ return self.scratch
192
+
193
+ def builtin_tools(self) -> tuple[str, ...]:
194
+ return ()
195
+
196
+ def servers(self) -> dict[str, Any]:
197
+ return {}
198
+
199
+ def briefing(self) -> str:
200
+ return (
201
+ "This is an externally hosted provider agent, not a repository. The sanitized "
202
+ "provider definition below is authoritative for its conversation, prompt, model, "
203
+ "voice, states, and tool schemas. There is no source code to search or open. Do not "
204
+ "invent behavior or tool inputs that are absent from this definition.\n\n"
205
+ f"PROVIDER DEFINITION:\n{json.dumps(self.profile, indent=2, sort_keys=True)}"
206
+ )
207
+
208
+
209
+ _REGISTRY: dict[str, Callable[..., AgentSource]] = {
210
+ "repo": lambda **kw: RepoSource(name=kw["name"], root=Path(kw["root"])),
211
+ "github": lambda **kw: GitHubSource(
212
+ name=kw["name"], root=Path(kw["root"]), url=kw.get("url", "")
213
+ ),
214
+ "spec": lambda **kw: SpecSource(
215
+ name=kw["name"],
216
+ system_prompt=kw.get("system_prompt", ""),
217
+ tool_schema=kw.get("tool_schema") or [],
218
+ data=kw.get("data") or {},
219
+ scratch=Path(kw.get("scratch", ".")),
220
+ ),
221
+ "provider": lambda **kw: ProviderSource(
222
+ name=kw["name"],
223
+ profile=kw.get("profile") or {},
224
+ scratch=Path(kw.get("scratch", ".")),
225
+ ),
226
+ }
227
+
228
+
229
+ def register_source(kind: str, factory: Callable[..., AgentSource]) -> None:
230
+ """Add a kind of agent. A provider connection is a class and one line here."""
231
+ _REGISTRY[kind] = factory
232
+
233
+
234
+ def resolve(kind: str, **kwargs: Any) -> AgentSource:
235
+ if kind not in _REGISTRY:
236
+ raise NotImplementedError(
237
+ f"no agent source of kind {kind!r}; registered kinds are "
238
+ f"{', '.join(sorted(_REGISTRY))}"
239
+ )
240
+ # An empty root used to resolve to the current directory, which is worse than failing: every
241
+ # later stage then reads a real path, finds the harness's own repository, and reports that the
242
+ # agent has no code on disk. Nothing downstream can tell that apart from an agent that really
243
+ # was given as a specification.
244
+ if "root" in kwargs and not str(kwargs.get("root") or "").strip():
245
+ raise ValueError(
246
+ f"a {kind!r} source needs the path its code lives at, and none was given. If this "
247
+ "agent has no code on disk, it is not this kind of source."
248
+ )
249
+ return _REGISTRY[kind](**kwargs)
250
+
251
+
252
+ def supported() -> tuple[str, ...]:
253
+ return tuple(sorted(_REGISTRY))
@@ -0,0 +1,140 @@
1
+ """What the harness itself spent, written where a deleted sandbox cannot take it with it.
2
+
3
+ Every model call the harness makes arrives at one place, `Stage`'s handling of `StageDone`, so the
4
+ ledger is fed from there rather than from each stage's own code: a writer added later is counted
5
+ without anybody remembering to count it. Parallel scenario writers and the suite review each open
6
+ their own session, which is why per-call-site accounting would have missed most of a large run's
7
+ spend.
8
+
9
+ The file is rewritten after every turn so the newest total is always on disk. A run that dies
10
+ mid-turn loses that turn only, and the platform reads the file while the sandbox is alive.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ import json
16
+ import os
17
+ import tempfile
18
+ from pathlib import Path
19
+ from typing import Any
20
+
21
+ JOURNAL_ALIAS = "ALK_SPEND_JOURNAL"
22
+
23
+ _stages: dict[str, dict[str, Any]] = {}
24
+ _path: Path | None = None
25
+
26
+
27
+ def journal_to(path: str | os.PathLike[str] | None) -> None:
28
+ """Where to keep the ledger. Nothing is written until this is set or the alias names a path."""
29
+ global _path
30
+ _path = Path(path) if path else None
31
+ if _path is not None:
32
+ _flush()
33
+
34
+
35
+ def _destination() -> Path | None:
36
+ if _path is not None:
37
+ return _path
38
+ named = os.environ.get(JOURNAL_ALIAS, "").strip()
39
+ return Path(named) if named else None
40
+
41
+
42
+ def record(
43
+ stage: str,
44
+ usd: float | None,
45
+ turns: int = 0,
46
+ models: set[str] | None = None,
47
+ tokens_in: int = 0,
48
+ tokens_out: int = 0,
49
+ tokens_cached: int = 0,
50
+ ) -> None:
51
+ """Add one session's reported spend. A backend that cannot price a call reports None.
52
+
53
+ ``tokens_cached`` is the part of ``tokens_in`` the provider served from its own cache. It is
54
+ reported rather than discounted, because the table here carries no cache rate and a guessed
55
+ one would be a made-up figure presented as a price. Carrying the count is what lets anyone
56
+ reconciling a bill see the size of the overstatement instead of inheriting it silently: two
57
+ reruns of the same authoring produced byte-identical ledgers four times apart in wall clock,
58
+ which is what caching looks like when nothing records it.
59
+ """
60
+ name = (stage or "stage").strip() or "stage"
61
+ entry = _stages.setdefault(
62
+ name,
63
+ {
64
+ "usd": 0.0,
65
+ "turns": 0,
66
+ "models": [],
67
+ "priced": 0,
68
+ "unpriced": 0,
69
+ "tokens_in": 0,
70
+ "tokens_out": 0,
71
+ "tokens_cached": 0,
72
+ },
73
+ )
74
+ if usd is None:
75
+ entry["unpriced"] += 1
76
+ else:
77
+ entry["usd"] = round(entry["usd"] + float(usd), 6)
78
+ entry["priced"] += 1
79
+ entry["turns"] += int(turns or 0)
80
+ entry["tokens_in"] += int(tokens_in or 0)
81
+ entry["tokens_out"] += int(tokens_out or 0)
82
+ entry["tokens_cached"] += int(tokens_cached or 0)
83
+ for model in sorted(models or set()):
84
+ if model not in entry["models"]:
85
+ entry["models"].append(model)
86
+ _flush()
87
+
88
+
89
+ def total_usd() -> float:
90
+ return round(sum(float(entry["usd"]) for entry in _stages.values()), 6)
91
+
92
+
93
+ def unpriced_turns() -> int:
94
+ """Turns whose backend reported no price. Nonzero means the total is a floor, not the answer."""
95
+ return sum(int(entry["unpriced"]) for entry in _stages.values())
96
+
97
+
98
+ def snapshot() -> dict[str, Any]:
99
+ return {
100
+ "schema": "futureagi.harness-spend.v1",
101
+ "total_usd": total_usd(),
102
+ "unpriced_turns": unpriced_turns(),
103
+ "stages": [
104
+ {
105
+ "stage": name,
106
+ **{
107
+ key: entry[key]
108
+ for key in (
109
+ "usd",
110
+ "turns",
111
+ "models",
112
+ "priced",
113
+ "unpriced",
114
+ "tokens_in",
115
+ "tokens_out",
116
+ "tokens_cached",
117
+ )
118
+ },
119
+ }
120
+ for name, entry in sorted(_stages.items())
121
+ ],
122
+ }
123
+
124
+
125
+ def _flush() -> None:
126
+ destination = _destination()
127
+ if destination is None:
128
+ return
129
+ try:
130
+ destination.parent.mkdir(parents=True, exist_ok=True)
131
+ # Written whole, then moved: a poll that reads mid-write must never see half a total.
132
+ handle = tempfile.NamedTemporaryFile(
133
+ "w", dir=destination.parent, prefix=".spend-", suffix=".json", delete=False
134
+ )
135
+ with handle as writing:
136
+ json.dump(snapshot(), writing, indent=2, sort_keys=True)
137
+ os.replace(handle.name, destination)
138
+ except OSError:
139
+ # Accounting must never be the reason a stage fails.
140
+ pass
@@ -0,0 +1,104 @@
1
+ """Bundle-owned observable HTTP tool proxy used by hosted process environments."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ import os
7
+ import time
8
+ from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
9
+ from urllib import error, request
10
+
11
+ import psycopg
12
+
13
+ PORT = int(os.environ["PORT"])
14
+ UPSTREAM = os.environ["UPSTREAM_URL"].rstrip("/")
15
+ DATABASE_URL = os.environ["DATABASE_URL"]
16
+
17
+
18
+ def _record(
19
+ name: str, arguments: object, result: object, ok: bool, failure: str = ""
20
+ ) -> None:
21
+ try:
22
+ with psycopg.connect(DATABASE_URL, autocommit=True) as connection:
23
+ connection.execute(
24
+ "INSERT INTO _alk_tool_trace(name, arguments, result, ok, error, at) "
25
+ "VALUES (%s, %s, %s, %s, %s, %s)",
26
+ (
27
+ name,
28
+ json.dumps(arguments),
29
+ json.dumps(result),
30
+ ok,
31
+ failure,
32
+ time.time(),
33
+ ),
34
+ )
35
+ except Exception:
36
+ # Evidence persistence must never alter the target tool response.
37
+ pass
38
+
39
+
40
+ class Handler(BaseHTTPRequestHandler):
41
+ def log_message(self, *_args: object) -> None:
42
+ return
43
+
44
+ def do_GET(self) -> None: # noqa: N802 - BaseHTTPRequestHandler API
45
+ if self.path == "/health":
46
+ self.send_response(200)
47
+ self.end_headers()
48
+ return
49
+ self._forward()
50
+
51
+ def do_POST(self) -> None: # noqa: N802 - BaseHTTPRequestHandler API
52
+ self._forward()
53
+
54
+ def _forward(self) -> None:
55
+ size = int(self.headers.get("content-length") or 0)
56
+ body = self.rfile.read(size) if size else b""
57
+ try:
58
+ arguments = json.loads(body) if body else {}
59
+ except ValueError:
60
+ arguments = {"_raw": body.decode("utf-8", errors="replace")}
61
+ outgoing = request.Request(
62
+ UPSTREAM + self.path,
63
+ data=body if self.command != "GET" else None,
64
+ method=self.command,
65
+ headers={
66
+ "content-type": self.headers.get("content-type", "application/json")
67
+ },
68
+ )
69
+ name = self.path.split("?", 1)[0].rstrip("/").rsplit("/", 1)[-1] or "unknown"
70
+ try:
71
+ with request.urlopen(outgoing, timeout=30) as response:
72
+ content = response.read()
73
+ status = int(response.status)
74
+ response_type = response.headers.get("content-type", "application/json")
75
+ try:
76
+ result = json.loads(content) if content else None
77
+ except ValueError:
78
+ result = content.decode("utf-8", errors="replace")
79
+ _record(name, arguments, result, status < 400)
80
+ self.send_response(status)
81
+ self.send_header("content-type", response_type)
82
+ self.send_header("content-length", str(len(content)))
83
+ self.end_headers()
84
+ self.wfile.write(content)
85
+ except error.HTTPError as exc:
86
+ content = exc.read()
87
+ failure = content.decode("utf-8", errors="replace")[:2000]
88
+ _record(name, arguments, None, False, failure)
89
+ self.send_response(exc.code)
90
+ self.send_header("content-length", str(len(content)))
91
+ self.end_headers()
92
+ self.wfile.write(content)
93
+ except Exception as exc:
94
+ _record(name, arguments, None, False, f"{type(exc).__name__}: unavailable")
95
+ content = json.dumps({"detail": "tool_upstream_unavailable"}).encode()
96
+ self.send_response(502)
97
+ self.send_header("content-type", "application/json")
98
+ self.send_header("content-length", str(len(content)))
99
+ self.end_headers()
100
+ self.wfile.write(content)
101
+
102
+
103
+ if __name__ == "__main__":
104
+ ThreadingHTTPServer(("127.0.0.1", PORT), Handler).serve_forever()