agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,294 @@
1
+ """Canonical ledger-row construction + serialization (Phase 8, ARCH ยง2a).
2
+
3
+ Imports: stdlib only plus the two reused live/ seams. The content address must
4
+ be byte-identical on any machine, so serialization replicates the
5
+ ``_schema.py:_json_sha256`` recipe exactly (``sort_keys=True``,
6
+ ``separators=(",", ":")``, ``default=str``) โ€” one canonicalization discipline,
7
+ no second divergable serializer (ARCH Decision 2). Redaction runs BEFORE the
8
+ row is content-addressed or written: the address is computed over redacted
9
+ bytes, so a re-run that re-redacts produces the same address.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ import hashlib
15
+ import json
16
+ import os
17
+ from typing import Any, Mapping, Sequence
18
+
19
+ from ..live._contract import AGENT_LEARNING_RUN_KIND, EVIDENCE_CLASSES, VERDICTS
20
+ from ..live._transcript import redact_env_values # the ONE redaction seam
21
+ from ._contract import (
22
+ LEDGER_ROW_SCHEMA,
23
+ NON_CANONICAL_FIELDS,
24
+ PHASES,
25
+ SEMCONV_VERSION_ENV,
26
+ )
27
+
28
+ # Fixed precision for floats in the addressed core โ€” the kit's existing
29
+ # rounding rule (live/_transcript.py:105 rounds to 6 places).
30
+ _FLOAT_PRECISION = 6
31
+
32
+
33
+ def canonical_row_bytes(row: Mapping[str, Any]) -> bytes:
34
+ """The exact bytes ``run_id`` is the SHA-256 of (the addressed core).
35
+
36
+ Excludes ``created_at``/``run_id``/``chain`` โ€” and ONLY those three
37
+ (ARCH ยง2a): wall-clock and the chain digest are envelope fields that must
38
+ never enter the content address.
39
+ """
40
+
41
+ preimage = {k: v for k, v in row.items() if k not in NON_CANONICAL_FIELDS}
42
+ return json.dumps(
43
+ preimage, sort_keys=True, separators=(",", ":"), default=str
44
+ ).encode("utf-8") # == _schema.py:_json_sha256 recipe, byte-identical
45
+
46
+
47
+ def canonical_row_address(row: Mapping[str, Any]) -> str:
48
+ """``run_id = SHA-256(canonical addressed core)`` (P8-D3)."""
49
+
50
+ return hashlib.sha256(canonical_row_bytes(row)).hexdigest()
51
+
52
+
53
+ def _redact_value(value: Any, required_env: Sequence[str]) -> Any:
54
+ """Walk a row value: redact env VALUES out of every string leaf and round
55
+ floats to fixed precision so the addressed core has no platform-variant
56
+ repr (ARCH ยง2a determinism rules)."""
57
+
58
+ if isinstance(value, str):
59
+ return redact_env_values(value, required_env)
60
+ if isinstance(value, bool):
61
+ return value
62
+ if isinstance(value, float):
63
+ return round(value, _FLOAT_PRECISION)
64
+ if isinstance(value, Mapping):
65
+ return {
66
+ _redact_value(key, required_env): _redact_value(item, required_env)
67
+ for key, item in value.items()
68
+ }
69
+ if isinstance(value, (list, tuple)):
70
+ return [_redact_value(item, required_env) for item in value]
71
+ return value
72
+
73
+
74
+ def build_ledger_row(
75
+ payload: Mapping[str, Any], *, required_env: Sequence[str] = ()
76
+ ) -> dict[str, Any]:
77
+ """Project an ``agent-learning.run.v1`` payload into a small ledger row of
78
+ metadata + content-addressed asset REFERENCES, never copies (PRD ยง4.1).
79
+
80
+ Redaction-before-serialize is the load-bearing ordering: ``_redact_value``
81
+ runs on the last step before ``canonical_row_address`` and before any disk
82
+ write โ€” the same seam+placement as ``live/_transcript.py:111``.
83
+ """
84
+
85
+ summary = payload.get("summary")
86
+ summary = summary if isinstance(summary, Mapping) else {}
87
+ evidence_class = payload.get("evidence_class")
88
+ if evidence_class not in EVIDENCE_CLASSES:
89
+ evidence_class = "local_gate" # absence => local_gate (BUILD ยง1.3)
90
+ capture = payload.get("capture")
91
+ capture = capture if isinstance(capture, Mapping) else {}
92
+ row: dict[str, Any] = {
93
+ "schema": LEDGER_ROW_SCHEMA,
94
+ "kind": AGENT_LEARNING_RUN_KIND, # always the canonical run kind
95
+ "phase": _infer_phase(payload),
96
+ "evidence_class": evidence_class,
97
+ "verdict": _project_verdict(payload, summary),
98
+ "scores": _project_scores(summary),
99
+ "gate_outcomes": _project_gate_outcomes(payload),
100
+ "semconv_version": os.environ.get(SEMCONV_VERSION_ENV) or "unset",
101
+ "manifest_address": _manifest_address(payload),
102
+ # ASSET REFERENCES โ€” content addresses, never copies (Rยง3.3):
103
+ "asset_refs": _asset_refs(payload),
104
+ "trace_ids": _trace_ids(payload),
105
+ "content_bearing": _content_bearing(payload, capture),
106
+ "redaction": _redaction_contract(capture),
107
+ }
108
+ # Redact env VALUES out of every string field BEFORE the row is
109
+ # content-addressed or written (Rยง1 2507.06350; PRD ยง4.1):
110
+ row = _redact_value(row, tuple(required_env))
111
+ row["run_id"] = canonical_row_address(row) # address AFTER redaction
112
+ return row # created_at/chain are added by the ledger append (envelope)
113
+
114
+
115
+ def content_admissible(run_payload: Mapping[str, Any]) -> bool:
116
+ """The content-sync admission predicate (PRD ยง4.2): the same
117
+ ``capture.redaction`` non-empty mapping + ``capture.reviewed is True``
118
+ shape the ``live_lane_boundary`` gate demands on captured fixtures."""
119
+
120
+ capture = run_payload.get("capture")
121
+ capture = capture if isinstance(capture, Mapping) else {}
122
+ redaction = capture.get("redaction")
123
+ has_map = isinstance(redaction, Mapping) and bool(redaction)
124
+ return has_map and capture.get("reviewed") is True
125
+
126
+
127
+ def declared_required_env(payload: Mapping[str, Any]) -> tuple[str, ...]:
128
+ """Collect declared env names from the run payload (names only โ€” the
129
+ redaction seam replaces their VALUES with ``[redacted:NAME]``)."""
130
+
131
+ names: list[str] = []
132
+ for source in (
133
+ payload.get("required_env"),
134
+ _mapping(payload.get("live_lane")).get("required_env"),
135
+ _mapping(payload.get("lane")).get("required_env"),
136
+ _mapping(_mapping(payload.get("capture")).get("redaction")),
137
+ ):
138
+ if isinstance(source, Mapping):
139
+ names.extend(str(name) for name in source)
140
+ elif isinstance(source, (list, tuple)):
141
+ names.extend(str(name) for name in source)
142
+ seen: dict[str, None] = {}
143
+ for name in names:
144
+ if name:
145
+ seen.setdefault(name, None)
146
+ return tuple(seen)
147
+
148
+
149
+ def _mapping(value: Any) -> dict[str, Any]:
150
+ return dict(value) if isinstance(value, Mapping) else {}
151
+
152
+
153
+ def _infer_phase(payload: Mapping[str, Any]) -> str:
154
+ explicit = payload.get("phase")
155
+ if isinstance(explicit, str) and explicit in PHASES:
156
+ return explicit
157
+ if isinstance(payload.get("live_lane"), Mapping) or isinstance(
158
+ payload.get("lane"), (str, Mapping)
159
+ ):
160
+ return "live"
161
+ if payload.get("optimization") is not None:
162
+ return "optimize"
163
+ if payload.get("redteam") is not None or payload.get("attacks") is not None:
164
+ return "redteam"
165
+ if payload.get("suite") is not None or payload.get("result_kinds") is not None:
166
+ return "suite"
167
+ if payload.get("evaluations") is not None or payload.get("evals") is not None:
168
+ return "evals"
169
+ return "simulate"
170
+
171
+
172
+ def _project_verdict(
173
+ payload: Mapping[str, Any], summary: Mapping[str, Any]
174
+ ) -> str | None:
175
+ """Echo the run's own verdict โ€” never recompute or reinterpret it
176
+ (ARCH ยง1.4: the ledger records the verdict it is handed)."""
177
+
178
+ for candidate in (payload.get("verdict"), summary.get("verdict")):
179
+ if isinstance(candidate, str) and candidate in VERDICTS:
180
+ return candidate
181
+ status = payload.get("status")
182
+ if status == "passed":
183
+ return "pass"
184
+ if status == "failed":
185
+ return "fail"
186
+ return None
187
+
188
+
189
+ def _project_scores(summary: Mapping[str, Any]) -> dict[str, float]:
190
+ scores: dict[str, float] = {}
191
+ for key, value in summary.items():
192
+ if isinstance(value, bool):
193
+ continue
194
+ if isinstance(value, (int, float)):
195
+ scores[str(key)] = round(float(value), _FLOAT_PRECISION)
196
+ return scores
197
+
198
+
199
+ def _project_gate_outcomes(payload: Mapping[str, Any]) -> dict[str, bool]:
200
+ outcomes: dict[str, bool] = {}
201
+ declared = payload.get("gate_outcomes")
202
+ if isinstance(declared, Mapping):
203
+ for key, value in declared.items():
204
+ outcomes[str(key)] = bool(value)
205
+ return outcomes
206
+ checks = payload.get("checks")
207
+ if isinstance(checks, (list, tuple)):
208
+ for check in checks:
209
+ if isinstance(check, Mapping) and check.get("id") is not None:
210
+ outcomes[str(check["id"])] = bool(
211
+ check.get("passed", check.get("status") == "passed")
212
+ )
213
+ return outcomes
214
+
215
+
216
+ def _manifest_address(payload: Mapping[str, Any]) -> str | None:
217
+ manifest = payload.get("manifest")
218
+ if isinstance(manifest, Mapping) and manifest:
219
+ data = json.dumps(
220
+ manifest, sort_keys=True, separators=(",", ":"), default=str
221
+ ).encode("utf-8")
222
+ return hashlib.sha256(data).hexdigest()
223
+ address = payload.get("manifest_address")
224
+ return str(address) if isinstance(address, str) and address else None
225
+
226
+
227
+ def _asset_refs(payload: Mapping[str, Any]) -> list[dict[str, Any]]:
228
+ refs: list[dict[str, Any]] = []
229
+ declared = payload.get("asset_refs")
230
+ if isinstance(declared, (list, tuple)):
231
+ for item in declared:
232
+ if not isinstance(item, Mapping):
233
+ continue
234
+ address = item.get("content_address") or item.get("content_hash")
235
+ if not address:
236
+ continue
237
+ ref: dict[str, Any] = {
238
+ "kind": str(item.get("kind") or "asset"),
239
+ "content_address": str(address),
240
+ }
241
+ if item.get("account_object_id"):
242
+ ref["account_object_id"] = str(item["account_object_id"])
243
+ refs.append(ref)
244
+ for plural, singular in (("personas", "persona"), ("scenarios", "scenario")):
245
+ for item in payload.get(plural) or []:
246
+ if not isinstance(item, Mapping):
247
+ continue
248
+ address = (
249
+ item.get("content_address")
250
+ or item.get("content_hash")
251
+ or item.get("version")
252
+ )
253
+ if not address:
254
+ continue
255
+ ref = {"kind": singular, "content_address": str(address)}
256
+ if item.get("account_object_id"):
257
+ ref["account_object_id"] = str(item["account_object_id"])
258
+ refs.append(ref)
259
+ return refs
260
+
261
+
262
+ def _trace_ids(payload: Mapping[str, Any]) -> list[str]:
263
+ declared = payload.get("trace_ids")
264
+ if isinstance(declared, (list, tuple)):
265
+ return [str(item) for item in declared if item]
266
+ return []
267
+
268
+
269
+ def _content_bearing(
270
+ payload: Mapping[str, Any], capture: Mapping[str, Any]
271
+ ) -> bool:
272
+ """True iff the row references captured content โ€” transcripts/prompts/
273
+ tool I/O (ARCH ยง2a); the sync content gate keys off it."""
274
+
275
+ if capture:
276
+ return True
277
+ if payload.get("transcripts"):
278
+ return True
279
+ declared = payload.get("asset_refs")
280
+ if isinstance(declared, (list, tuple)):
281
+ for item in declared:
282
+ if isinstance(item, Mapping) and item.get("kind") == "transcript":
283
+ return True
284
+ return False
285
+
286
+
287
+ def _redaction_contract(capture: Mapping[str, Any]) -> dict[str, Any] | None:
288
+ """The capture+redaction mapping for content-bearing rows: env NAMES +
289
+ strategy โ€” names always, values never. ``None`` on metadata-only rows."""
290
+
291
+ redaction = capture.get("redaction")
292
+ if isinstance(redaction, Mapping) and redaction:
293
+ return {str(name): str(strategy) for name, strategy in redaction.items()}
294
+ return None
@@ -0,0 +1,233 @@
1
+ """``run_telemetry`` (Phase 14, ARCH ยง4) โ€” the ONE telemetry surface every kit run
2
+ wraps its body in. The W&B / promptfoo model:
3
+
4
+ * local path โ€” ALWAYS: append the ledger row + return/print a RunSummary. No
5
+ network. Works credential-free (promptfoo-local).
6
+ * cloud path โ€” ADDITIVE, only when keys resolve and the collector is reachable:
7
+ emit the run as a real trace and print a clickable dashboard URL (W&B
8
+ "View run at โ€ฆ"). The URL is printed ONLY on an OBSERVED export success.
9
+
10
+ Logs go to STDERR (W&B convention) so a kit run's STDOUT stays clean (the gate
11
+ example asserts empty stdout). Keys gate the destination, never the capability
12
+ (P8-D2); ``AGENT_LEARNING_TELEMETRY=off`` binds everything (P8-D6).
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ import sys
18
+ from contextlib import contextmanager
19
+ from dataclasses import dataclass, field
20
+ from typing import Any, Iterator, Mapping
21
+
22
+ from ..config import AgentLearningConfig
23
+ from ._contract import SYNC_MODE_AUTO, kill_switch_on, ledger_dir, sync_mode
24
+ from ._ledger import RunLedger
25
+ from ._row import build_ledger_row
26
+
27
+
28
+ @dataclass
29
+ class RunSummary:
30
+ """What a kit run reports โ€” the value the caller attaches to its result and
31
+ ``_log_summary`` renders."""
32
+
33
+ kind: str
34
+ name: str
35
+ status: str = "local" # local | synced | export_failed | deferred | off
36
+ metrics: dict[str, Any] = field(default_factory=dict)
37
+ run_id: str | None = None
38
+ trace_id: str | None = None
39
+ dashboard_url: str | None = None
40
+ url_kind: str | None = None
41
+ reason: str | None = None
42
+
43
+ def as_dict(self) -> dict[str, Any]:
44
+ return {
45
+ "kind": self.kind,
46
+ "name": self.name,
47
+ "status": self.status,
48
+ "metrics": dict(self.metrics),
49
+ "run_id": self.run_id,
50
+ "trace_id": self.trace_id,
51
+ "dashboard_url": self.dashboard_url,
52
+ "url_kind": self.url_kind,
53
+ "reason": self.reason,
54
+ }
55
+
56
+
57
+ class RunRecorder:
58
+ """Collects metrics + child-span specs during a run; ``summary`` is populated
59
+ on context exit and readable by the caller afterwards."""
60
+
61
+ def __init__(self, kind: str, name: str, world: str | None = None) -> None:
62
+ self.kind = kind
63
+ self.name = name
64
+ self.world = world
65
+ self.metrics: dict[str, Any] = {}
66
+ self.children: list[tuple[str, dict[str, Any]]] = []
67
+ self.verdict: str | None = None
68
+ self.summary: RunSummary | None = None
69
+
70
+ def set_metrics(self, **metrics: Any) -> None:
71
+ self.metrics.update(metrics)
72
+
73
+ def set_verdict(self, verdict: str) -> None:
74
+ self.verdict = verdict
75
+
76
+ def add_child(self, name: str, attrs: Mapping[str, Any]) -> None:
77
+ self.children.append((str(name), dict(attrs)))
78
+
79
+
80
+ def _ledger_payload(rec: RunRecorder) -> dict[str, Any]:
81
+ return {
82
+ "kind": rec.kind,
83
+ "summary": {
84
+ "name": rec.name,
85
+ "verdict": rec.verdict,
86
+ "metrics": rec.metrics,
87
+ },
88
+ }
89
+
90
+
91
+ def _log_summary(summary: RunSummary) -> None:
92
+ """Render the W&B / promptfoo line to stderr."""
93
+
94
+ metric_bits = " ยท ".join(
95
+ f"{k} {v}" for k, v in summary.metrics.items()
96
+ if isinstance(v, (int, float, str))
97
+ )
98
+ head = f"agent-learning: {summary.kind} '{summary.name}'"
99
+ if metric_bits:
100
+ head += f" ยท {metric_bits}"
101
+ print(head, file=sys.stderr)
102
+
103
+ if summary.status == "synced" and summary.dashboard_url:
104
+ if summary.url_kind == "deep_link":
105
+ print(f"agent-learning: ๐Ÿ”— view in dashboard โ†’ {summary.dashboard_url}", file=sys.stderr)
106
+ elif summary.url_kind == "project":
107
+ print(f"agent-learning: ๐Ÿ”— view in dashboard (project) โ†’ {summary.dashboard_url}", file=sys.stderr)
108
+ else: # list_fallback
109
+ print(
110
+ f"agent-learning: ๐Ÿ”— dashboard โ†’ {summary.dashboard_url} "
111
+ f"(find project '{summary.name}' / '{summary.metrics.get('project_name', '')}')",
112
+ file=sys.stderr,
113
+ )
114
+ elif summary.status == "export_failed":
115
+ print(
116
+ f"agent-learning: โš  dashboard export not accepted ({summary.reason}) โ€” logged locally only",
117
+ file=sys.stderr,
118
+ )
119
+ elif summary.status == "deferred":
120
+ print(
121
+ f"agent-learning: dashboard unreachable ({summary.reason}) โ€” logged locally only",
122
+ file=sys.stderr,
123
+ )
124
+ elif summary.status == "off":
125
+ print("agent-learning: telemetry off โ€” nothing logged or sent", file=sys.stderr)
126
+ else: # local
127
+ print(f"agent-learning: logged locally โ†’ {ledger_dir()}", file=sys.stderr)
128
+ print(
129
+ "agent-learning: set FI_API_KEY + FI_SECRET_KEY to view runs in the dashboard",
130
+ file=sys.stderr,
131
+ )
132
+
133
+
134
+ def _finalize(rec: RunRecorder, *, project_name: str | None) -> RunSummary:
135
+ summary = RunSummary(kind=rec.kind, name=rec.name, metrics=dict(rec.metrics))
136
+
137
+ # Kill switch binds EVERYTHING incl. the ledger (P8-D6).
138
+ if kill_switch_on():
139
+ summary.status = "off"
140
+ return summary
141
+
142
+ # Local path โ€” always (FR2).
143
+ row = build_ledger_row(_ledger_payload(rec))
144
+ try:
145
+ RunLedger().append(row)
146
+ except Exception: # noqa: BLE001 โ€” a ledger write failure must not break the run
147
+ pass
148
+ summary.run_id = row.get("run_id")
149
+
150
+ config = AgentLearningConfig.from_env()
151
+ # Cloud path requires keys AND mode=auto (W&B-online). mode=local (the test/
152
+ # gate default) queues locally โ€” no surprise network in release/CI (P8).
153
+ if not (config.api_key and config.secret_key) or sync_mode() != SYNC_MODE_AUTO:
154
+ summary.status = "local"
155
+ return summary
156
+
157
+ # Cloud path โ€” additive (FR3). Import network-capable code only here.
158
+ from . import _emit, _url
159
+
160
+ proj = project_name or "agent-learning"
161
+ summary.metrics.setdefault("project_name", proj)
162
+ emit = _emit.keyed_emit(
163
+ span_name=_emit.RUN_SPAN_NAME,
164
+ root_attrs={"kind": rec.kind, "name": rec.name, **rec.metrics},
165
+ children=rec.children,
166
+ project_name=proj,
167
+ headers={"X-Api-Key": config.api_key, "X-Secret-Key": config.secret_key},
168
+ run_id=summary.run_id or "",
169
+ phase=rec.kind,
170
+ world=rec.world,
171
+ )
172
+ summary.status = emit["status"]
173
+ summary.reason = emit.get("reason")
174
+ if emit["status"] == "synced":
175
+ summary.trace_id = emit.get("trace_id")
176
+ url = _url.build_dashboard_url(proj, summary.trace_id, config=config)
177
+ summary.dashboard_url = url["url"]
178
+ summary.url_kind = url["kind"]
179
+ return summary
180
+
181
+
182
+ def emit_run(
183
+ *,
184
+ kind: str,
185
+ name: str,
186
+ metrics: Mapping[str, Any] | None = None,
187
+ verdict: str | None = None,
188
+ children: list[tuple[str, dict[str, Any]]] | None = None,
189
+ world: str | None = None,
190
+ project_name: str | None = None,
191
+ ) -> RunSummary:
192
+ """Non-context entrypoint for code paths that have already computed their
193
+ results (``run_benchmark`` / ``optimize_against_dataset`` / ``improve_agent_
194
+ code``): finalize one run (local ledger always; cloud emit when mode=auto +
195
+ keys), log the W&B/promptfoo line, and return the summary. Never raises into
196
+ the caller."""
197
+
198
+ try:
199
+ rec = RunRecorder(kind=kind, name=name, world=world)
200
+ if metrics:
201
+ rec.set_metrics(**dict(metrics))
202
+ if verdict:
203
+ rec.set_verdict(verdict)
204
+ for child_name, attrs in children or []:
205
+ rec.add_child(child_name, attrs)
206
+ rec.summary = _finalize(rec, project_name=project_name)
207
+ _log_summary(rec.summary)
208
+ return rec.summary
209
+ except Exception: # noqa: BLE001 โ€” telemetry is a side-channel, never fatal
210
+ return RunSummary(kind=kind, name=name, status="local")
211
+
212
+
213
+ @contextmanager
214
+ def run_telemetry(
215
+ *,
216
+ kind: str,
217
+ name: str,
218
+ world: str | None = None,
219
+ project_name: str | None = None,
220
+ ) -> Iterator[RunRecorder]:
221
+ """Wrap a kit run. Yields a ``RunRecorder``; after the block, ``rec.summary``
222
+ holds the finalized ``RunSummary`` (status + dashboard URL). Telemetry never
223
+ raises into the wrapped run."""
224
+
225
+ rec = RunRecorder(kind=kind, name=name, world=world)
226
+ try:
227
+ yield rec
228
+ finally:
229
+ try:
230
+ rec.summary = _finalize(rec, project_name=project_name)
231
+ _log_summary(rec.summary)
232
+ except Exception: # noqa: BLE001 โ€” telemetry is a side-channel, never fatal
233
+ rec.summary = RunSummary(kind=kind, name=name, status="local")