agent-learning-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (642) hide show
  1. agent_learning_kit-0.1.0.dist-info/METADATA +381 -0
  2. agent_learning_kit-0.1.0.dist-info/RECORD +642 -0
  3. agent_learning_kit-0.1.0.dist-info/WHEEL +4 -0
  4. agent_learning_kit-0.1.0.dist-info/entry_points.txt +5 -0
  5. agent_learning_kit-0.1.0.dist-info/licenses/LICENSE +173 -0
  6. agent_learning_kit-0.1.0.dist-info/licenses/NOTICE +7 -0
  7. fi/__init__.py +5 -0
  8. fi/alk/__init__.py +57 -0
  9. fi/alk/_facade.py +31 -0
  10. fi/alk/_module_alias.py +68 -0
  11. fi/alk/_paths.py +14 -0
  12. fi/alk/_schema.py +522 -0
  13. fi/alk/actions.py +727 -0
  14. fi/alk/bench/__init__.py +517 -0
  15. fi/alk/bench/_codeexec.py +213 -0
  16. fi/alk/bench/_coding.py +215 -0
  17. fi/alk/bench/_docker.py +237 -0
  18. fi/alk/bench/_grader.py +286 -0
  19. fi/alk/bench/_pull.py +212 -0
  20. fi/alk/bench/_voice.py +147 -0
  21. fi/alk/capabilities.py +627 -0
  22. fi/alk/cli.py +6396 -0
  23. fi/alk/config.py +130 -0
  24. fi/alk/cua_loop.py +562 -0
  25. fi/alk/evals.py +2351 -0
  26. fi/alk/extensions.py +163 -0
  27. fi/alk/harness/ARCHITECTURE.md +231 -0
  28. fi/alk/harness/DESIGN.md +246 -0
  29. fi/alk/harness/ENVIRONMENT_CONFORMANCE.md +127 -0
  30. fi/alk/harness/HOW-IT-WORKS.md +297 -0
  31. fi/alk/harness/IMPLEMENTATION_AND_VALIDATION_STATUS.md +229 -0
  32. fi/alk/harness/README.md +417 -0
  33. fi/alk/harness/__init__.py +77 -0
  34. fi/alk/harness/__main__.py +3 -0
  35. fi/alk/harness/amend.py +312 -0
  36. fi/alk/harness/artifacts.py +319 -0
  37. fi/alk/harness/authoring_entrypoint.py +189 -0
  38. fi/alk/harness/authoring_runtime_validation.py +267 -0
  39. fi/alk/harness/backends/README.md +43 -0
  40. fi/alk/harness/backends/__init__.py +122 -0
  41. fi/alk/harness/backends/base.py +241 -0
  42. fi/alk/harness/backends/claude.py +211 -0
  43. fi/alk/harness/backends/files.py +182 -0
  44. fi/alk/harness/backends/vertex_gemini.py +457 -0
  45. fi/alk/harness/background_noise.py +95 -0
  46. fi/alk/harness/build.py +385 -0
  47. fi/alk/harness/bundle.py +593 -0
  48. fi/alk/harness/bundle_author_v2.py +1831 -0
  49. fi/alk/harness/bundle_v2.py +719 -0
  50. fi/alk/harness/call_runner.py +1440 -0
  51. fi/alk/harness/callback_http_adapter.py +111 -0
  52. fi/alk/harness/catalogue.py +287 -0
  53. fi/alk/harness/chat.py +428 -0
  54. fi/alk/harness/chat_call_runner.py +506 -0
  55. fi/alk/harness/checks.py +136 -0
  56. fi/alk/harness/cli.py +1354 -0
  57. fi/alk/harness/config.py +338 -0
  58. fi/alk/harness/contract.py +718 -0
  59. fi/alk/harness/credentials.py +674 -0
  60. fi/alk/harness/data/persona_vocabulary.json +111 -0
  61. fi/alk/harness/environment.py +99 -0
  62. fi/alk/harness/environment_plan.py +168 -0
  63. fi/alk/harness/events.py +125 -0
  64. fi/alk/harness/executor.py +304 -0
  65. fi/alk/harness/folder.py +234 -0
  66. fi/alk/harness/generated_runtime.py +815 -0
  67. fi/alk/harness/github.py +72 -0
  68. fi/alk/harness/hosted_authoring_entrypoint.py +183 -0
  69. fi/alk/harness/hosted_entrypoint.py +2402 -0
  70. fi/alk/harness/hosted_scheduler.py +2218 -0
  71. fi/alk/harness/job.py +426 -0
  72. fi/alk/harness/judge.py +184 -0
  73. fi/alk/harness/livekit_source.py +50 -0
  74. fi/alk/harness/livekit_tool_trace_bootstrap.py +71 -0
  75. fi/alk/harness/observability.py +208 -0
  76. fi/alk/harness/outbound.py +3252 -0
  77. fi/alk/harness/packaging.py +515 -0
  78. fi/alk/harness/persona_guides.py +157 -0
  79. fi/alk/harness/platform.py +692 -0
  80. fi/alk/harness/process_preflight.py +764 -0
  81. fi/alk/harness/process_runtime.py +5670 -0
  82. fi/alk/harness/prove.py +425 -0
  83. fi/alk/harness/provider_import.py +703 -0
  84. fi/alk/harness/provider_lifecycle.py +392 -0
  85. fi/alk/harness/provision.py +2896 -0
  86. fi/alk/harness/reception.py +147 -0
  87. fi/alk/harness/retell_chat_call_runner.py +373 -0
  88. fi/alk/harness/run/__init__.py +296 -0
  89. fi/alk/harness/run/alk.py +184 -0
  90. fi/alk/harness/run/call.py +162 -0
  91. fi/alk/harness/run/conversation.py +264 -0
  92. fi/alk/harness/run/data/voices_by_language_and_gender.json +693 -0
  93. fi/alk/harness/run/evidence.py +195 -0
  94. fi/alk/harness/run/grade.py +598 -0
  95. fi/alk/harness/run/live.py +297 -0
  96. fi/alk/harness/run/models.py +56 -0
  97. fi/alk/harness/run/platform_evals.py +227 -0
  98. fi/alk/harness/run/sdk_voice.py +130 -0
  99. fi/alk/harness/run/simulation.py +1209 -0
  100. fi/alk/harness/run/stage.py +91 -0
  101. fi/alk/harness/run/targets.py +508 -0
  102. fi/alk/harness/run/tools.py +601 -0
  103. fi/alk/harness/run/voice.py +340 -0
  104. fi/alk/harness/runtime.py +172 -0
  105. fi/alk/harness/sandbox_server.py +2011 -0
  106. fi/alk/harness/sandbox_worker.py +44 -0
  107. fi/alk/harness/scenario.py +1048 -0
  108. fi/alk/harness/scenario_source.py +879 -0
  109. fi/alk/harness/scenario_tools.py +1143 -0
  110. fi/alk/harness/scenarios.py +915 -0
  111. fi/alk/harness/secrets.py +168 -0
  112. fi/alk/harness/service_catalog.py +97 -0
  113. fi/alk/harness/session.py +391 -0
  114. fi/alk/harness/sessions.py +372 -0
  115. fi/alk/harness/simulator.py +76 -0
  116. fi/alk/harness/simulator_voice.py +928 -0
  117. fi/alk/harness/skills/build-environment/SKILL.md +538 -0
  118. fi/alk/harness/skills/harness.md +131 -0
  119. fi/alk/harness/skills/kinds/chat.md +48 -0
  120. fi/alk/harness/skills/kinds/voice-voicemail.md +63 -0
  121. fi/alk/harness/skills/kinds/voice.md +59 -0
  122. fi/alk/harness/skills/plan-suite/SKILL.md +103 -0
  123. fi/alk/harness/skills/provision-environment/SKILL.md +136 -0
  124. fi/alk/harness/skills/run-scenarios/SKILL.md +112 -0
  125. fi/alk/harness/skills/understand-agent/SKILL.md +251 -0
  126. fi/alk/harness/skills/write-scenarios/SKILL.md +606 -0
  127. fi/alk/harness/skills/write-scenarios/references/refusals.md +28 -0
  128. fi/alk/harness/skills/write-scenarios/references/world-api.md +92 -0
  129. fi/alk/harness/source_data_invariants.py +444 -0
  130. fi/alk/harness/source_tool_evidence.py +79 -0
  131. fi/alk/harness/sources.py +253 -0
  132. fi/alk/harness/spend.py +140 -0
  133. fi/alk/harness/tool_trace_proxy.py +104 -0
  134. fi/alk/harness/tools.py +1018 -0
  135. fi/alk/harness/understand.py +169 -0
  136. fi/alk/harness/voicemail_audio.py +74 -0
  137. fi/alk/harness/world/__init__.py +33 -0
  138. fi/alk/harness/world/errors.py +68 -0
  139. fi/alk/harness/world/expectations.py +91 -0
  140. fi/alk/harness/world/handle.py +538 -0
  141. fi/alk/harness/world/kinds.py +196 -0
  142. fi/alk/harness/world/mutate.py +186 -0
  143. fi/alk/harness/world/probe.py +413 -0
  144. fi/alk/harness/world/provision.py +511 -0
  145. fi/alk/harness/world/provisioned.py +191 -0
  146. fi/alk/harness/world/runtime.py +616 -0
  147. fi/alk/harness/world/snapshot.py +288 -0
  148. fi/alk/harness/world/stores/__init__.py +305 -0
  149. fi/alk/harness/world/stores/container.py +215 -0
  150. fi/alk/harness/world/stores/inprocess.py +346 -0
  151. fi/alk/harness/world/stores/postgres.py +481 -0
  152. fi/alk/harness/world/stores/prove.py +202 -0
  153. fi/alk/harness/world/stores/sqlite.py +245 -0
  154. fi/alk/harness/world/stores/written.py +182 -0
  155. fi/alk/harness/world/tools.py +1516 -0
  156. fi/alk/harness/world/workspace.py +144 -0
  157. fi/alk/image_loop.py +453 -0
  158. fi/alk/image_perturb.py +241 -0
  159. fi/alk/improve.py +274 -0
  160. fi/alk/live/__init__.py +154 -0
  161. fi/alk/live/_attribution.py +184 -0
  162. fi/alk/live/_capture.py +264 -0
  163. fi/alk/live/_codec.py +391 -0
  164. fi/alk/live/_contract.py +134 -0
  165. fi/alk/live/_loopback.py +316 -0
  166. fi/alk/live/_perturb.py +449 -0
  167. fi/alk/live/_runner.py +386 -0
  168. fi/alk/live/_stats.py +561 -0
  169. fi/alk/live/_transcript.py +240 -0
  170. fi/alk/live/_workers/__init__.py +9 -0
  171. fi/alk/live/_workers/a2a_worker.py +316 -0
  172. fi/alk/live/_workers/langgraph_worker.py +217 -0
  173. fi/alk/live/_workers/livekit_worker.py +207 -0
  174. fi/alk/live/_workers/mcp_loopback_server.py +46 -0
  175. fi/alk/live/_workers/mcp_worker.py +158 -0
  176. fi/alk/live/_workers/pipecat_worker.py +189 -0
  177. fi/alk/live/a2a_lane.py +138 -0
  178. fi/alk/live/langgraph_lane.py +339 -0
  179. fi/alk/live/livekit_lane.py +376 -0
  180. fi/alk/live/mcp_lane.py +172 -0
  181. fi/alk/live/pipecat_lane.py +341 -0
  182. fi/alk/live/voice_redteam.py +494 -0
  183. fi/alk/loss.py +306 -0
  184. fi/alk/optimize.py +36260 -0
  185. fi/alk/practice/__init__.py +51 -0
  186. fi/alk/practice/_assess.py +103 -0
  187. fi/alk/practice/_budget.py +81 -0
  188. fi/alk/practice/_calibrate.py +69 -0
  189. fi/alk/practice/_capstone.py +86 -0
  190. fi/alk/practice/_contract.py +91 -0
  191. fi/alk/practice/_diagnose.py +79 -0
  192. fi/alk/practice/_drill.py +196 -0
  193. fi/alk/practice/_experiment.py +720 -0
  194. fi/alk/practice/_schedule.py +102 -0
  195. fi/alk/practice/_store.py +194 -0
  196. fi/alk/practice/_trainer.py +245 -0
  197. fi/alk/practice/_update.py +125 -0
  198. fi/alk/redteam.py +2621 -0
  199. fi/alk/rewardhack.py +237 -0
  200. fi/alk/simulate.py +10351 -0
  201. fi/alk/studio/__init__.py +82 -0
  202. fi/alk/studio/_bias.py +314 -0
  203. fi/alk/studio/_calibration.py +522 -0
  204. fi/alk/studio/_coverage.py +262 -0
  205. fi/alk/studio/_download.py +665 -0
  206. fi/alk/studio/_fidelity_attack.py +114 -0
  207. fi/alk/studio/_generate.py +652 -0
  208. fi/alk/studio/_library.py +370 -0
  209. fi/alk/studio/_scan.py +134 -0
  210. fi/alk/studio/_upgrade.py +42 -0
  211. fi/alk/studio/_vendor.py +172 -0
  212. fi/alk/suite.py +4200 -0
  213. fi/alk/tasks.py +828 -0
  214. fi/alk/telemetry/__init__.py +149 -0
  215. fi/alk/telemetry/_contract.py +141 -0
  216. fi/alk/telemetry/_emit.py +182 -0
  217. fi/alk/telemetry/_ledger.py +296 -0
  218. fi/alk/telemetry/_queue.py +127 -0
  219. fi/alk/telemetry/_row.py +294 -0
  220. fi/alk/telemetry/_run.py +233 -0
  221. fi/alk/telemetry/_sync.py +193 -0
  222. fi/alk/telemetry/_url.py +119 -0
  223. fi/alk/trinity.py +49397 -0
  224. fi/alk/voice_loop.py +174 -0
  225. fi/api/__init__.py +1 -0
  226. fi/api/auth.py +137 -0
  227. fi/api/types.py +29 -0
  228. fi/cli/__init__.py +9 -0
  229. fi/cli/assertions/__init__.py +25 -0
  230. fi/cli/assertions/conditions.py +76 -0
  231. fi/cli/assertions/evaluator.py +286 -0
  232. fi/cli/assertions/exit_codes.py +20 -0
  233. fi/cli/assertions/parser.py +131 -0
  234. fi/cli/assertions/reporter.py +194 -0
  235. fi/cli/commands/__init__.py +9 -0
  236. fi/cli/commands/config.py +165 -0
  237. fi/cli/commands/export.py +208 -0
  238. fi/cli/commands/init.py +112 -0
  239. fi/cli/commands/list_cmd.py +213 -0
  240. fi/cli/commands/run.py +486 -0
  241. fi/cli/commands/validate.py +173 -0
  242. fi/cli/commands/view.py +424 -0
  243. fi/cli/config/__init__.py +6 -0
  244. fi/cli/config/defaults.py +206 -0
  245. fi/cli/config/loader.py +155 -0
  246. fi/cli/config/schema.py +174 -0
  247. fi/cli/main.py +78 -0
  248. fi/cli/output/__init__.py +6 -0
  249. fi/cli/output/formatters.py +106 -0
  250. fi/cli/output/reporters.py +46 -0
  251. fi/cli/storage/__init__.py +5 -0
  252. fi/cli/storage/run_history.py +249 -0
  253. fi/cli/utils/__init__.py +5 -0
  254. fi/cli/utils/console.py +44 -0
  255. fi/evals/__init__.py +131 -0
  256. fi/evals/autoeval/__init__.py +137 -0
  257. fi/evals/autoeval/analyzer.py +211 -0
  258. fi/evals/autoeval/config.py +244 -0
  259. fi/evals/autoeval/export.py +213 -0
  260. fi/evals/autoeval/interactive.py +283 -0
  261. fi/evals/autoeval/pipeline.py +625 -0
  262. fi/evals/autoeval/prompts.py +139 -0
  263. fi/evals/autoeval/recommender.py +242 -0
  264. fi/evals/autoeval/rules.py +589 -0
  265. fi/evals/autoeval/templates.py +299 -0
  266. fi/evals/autoeval/types.py +232 -0
  267. fi/evals/core/__init__.py +16 -0
  268. fi/evals/core/cloud_registry.py +184 -0
  269. fi/evals/core/engines.py +368 -0
  270. fi/evals/core/evaluate.py +319 -0
  271. fi/evals/core/judge_prompt.py +90 -0
  272. fi/evals/core/prompt_generator.py +83 -0
  273. fi/evals/core/registry.py +57 -0
  274. fi/evals/core/result.py +55 -0
  275. fi/evals/evaluator.py +721 -0
  276. fi/evals/execution.py +168 -0
  277. fi/evals/feedback/__init__.py +32 -0
  278. fi/evals/feedback/calibrator.py +160 -0
  279. fi/evals/feedback/collector.py +214 -0
  280. fi/evals/feedback/hooks.py +81 -0
  281. fi/evals/feedback/retriever.py +128 -0
  282. fi/evals/feedback/store.py +272 -0
  283. fi/evals/feedback/types.py +99 -0
  284. fi/evals/framework/README.md +79 -0
  285. fi/evals/framework/__init__.py +267 -0
  286. fi/evals/framework/backends/Dockerfile.eval-runner +33 -0
  287. fi/evals/framework/backends/__init__.py +99 -0
  288. fi/evals/framework/backends/_container.py +141 -0
  289. fi/evals/framework/backends/_utils.py +145 -0
  290. fi/evals/framework/backends/base.py +223 -0
  291. fi/evals/framework/backends/celery_backend.py +417 -0
  292. fi/evals/framework/backends/celery_worker.py +78 -0
  293. fi/evals/framework/backends/kubernetes_backend.py +665 -0
  294. fi/evals/framework/backends/ray_backend.py +521 -0
  295. fi/evals/framework/backends/temporal.py +350 -0
  296. fi/evals/framework/backends/temporal_worker.py +126 -0
  297. fi/evals/framework/backends/thread_pool.py +286 -0
  298. fi/evals/framework/context.py +258 -0
  299. fi/evals/framework/enrichment.py +306 -0
  300. fi/evals/framework/evals/__init__.py +68 -0
  301. fi/evals/framework/evals/agentic.py +399 -0
  302. fi/evals/framework/evals/builder.py +609 -0
  303. fi/evals/framework/evals/semantic.py +142 -0
  304. fi/evals/framework/evaluator.py +647 -0
  305. fi/evals/framework/evaluators/__init__.py +22 -0
  306. fi/evals/framework/evaluators/blocking.py +347 -0
  307. fi/evals/framework/evaluators/non_blocking.py +577 -0
  308. fi/evals/framework/propagation.py +421 -0
  309. fi/evals/framework/protocols.py +385 -0
  310. fi/evals/framework/registry.py +370 -0
  311. fi/evals/framework/resilience/__init__.py +150 -0
  312. fi/evals/framework/resilience/circuit_breaker.py +309 -0
  313. fi/evals/framework/resilience/degradation.py +355 -0
  314. fi/evals/framework/resilience/health.py +505 -0
  315. fi/evals/framework/resilience/rate_limiter.py +228 -0
  316. fi/evals/framework/resilience/retry.py +274 -0
  317. fi/evals/framework/resilience/types.py +288 -0
  318. fi/evals/framework/resilience/wrapper.py +433 -0
  319. fi/evals/framework/types.py +218 -0
  320. fi/evals/guardrails/README.md +915 -0
  321. fi/evals/guardrails/__init__.py +96 -0
  322. fi/evals/guardrails/backends/__init__.py +43 -0
  323. fi/evals/guardrails/backends/azure.py +361 -0
  324. fi/evals/guardrails/backends/base.py +88 -0
  325. fi/evals/guardrails/backends/generic_llm.py +163 -0
  326. fi/evals/guardrails/backends/granite.py +216 -0
  327. fi/evals/guardrails/backends/llamaguard.py +221 -0
  328. fi/evals/guardrails/backends/local_base.py +479 -0
  329. fi/evals/guardrails/backends/openai.py +365 -0
  330. fi/evals/guardrails/backends/qwen.py +170 -0
  331. fi/evals/guardrails/backends/shieldgemma.py +154 -0
  332. fi/evals/guardrails/backends/turing.py +235 -0
  333. fi/evals/guardrails/backends/vllm_client.py +321 -0
  334. fi/evals/guardrails/backends/wildguard.py +188 -0
  335. fi/evals/guardrails/base.py +888 -0
  336. fi/evals/guardrails/config.py +221 -0
  337. fi/evals/guardrails/discovery.py +243 -0
  338. fi/evals/guardrails/gateway.py +437 -0
  339. fi/evals/guardrails/registry.py +231 -0
  340. fi/evals/guardrails/scanners/__init__.py +127 -0
  341. fi/evals/guardrails/scanners/base.py +191 -0
  342. fi/evals/guardrails/scanners/code_injection.py +243 -0
  343. fi/evals/guardrails/scanners/eval_delegate.py +574 -0
  344. fi/evals/guardrails/scanners/invisible_chars.py +351 -0
  345. fi/evals/guardrails/scanners/jailbreak.py +412 -0
  346. fi/evals/guardrails/scanners/language.py +288 -0
  347. fi/evals/guardrails/scanners/pipeline.py +260 -0
  348. fi/evals/guardrails/scanners/regex.py +311 -0
  349. fi/evals/guardrails/scanners/secrets.py +274 -0
  350. fi/evals/guardrails/scanners/topics.py +649 -0
  351. fi/evals/guardrails/scanners/urls.py +341 -0
  352. fi/evals/guardrails/types.py +96 -0
  353. fi/evals/llm/__init__.py +3 -0
  354. fi/evals/llm/base_llm_provider.py +35 -0
  355. fi/evals/llm/providers/litellm.py +70 -0
  356. fi/evals/local/__init__.py +90 -0
  357. fi/evals/local/evaluator.py +690 -0
  358. fi/evals/local/execution_mode.py +121 -0
  359. fi/evals/local/llm.py +489 -0
  360. fi/evals/local/metrics/__init__.py +19 -0
  361. fi/evals/local/registry.py +360 -0
  362. fi/evals/manager.py +1018 -0
  363. fi/evals/manager_types.py +362 -0
  364. fi/evals/metrics/__init__.py +185 -0
  365. fi/evals/metrics/agents/__init__.py +74 -0
  366. fi/evals/metrics/agents/metrics.py +693 -0
  367. fi/evals/metrics/agents/report.py +36463 -0
  368. fi/evals/metrics/agents/types.py +160 -0
  369. fi/evals/metrics/base_llm_metric.py +111 -0
  370. fi/evals/metrics/base_metric.py +138 -0
  371. fi/evals/metrics/code_security/__init__.py +305 -0
  372. fi/evals/metrics/code_security/analyzer.py +985 -0
  373. fi/evals/metrics/code_security/benchmarks/__init__.py +73 -0
  374. fi/evals/metrics/code_security/benchmarks/builtin.py +750 -0
  375. fi/evals/metrics/code_security/benchmarks/loader.py +580 -0
  376. fi/evals/metrics/code_security/benchmarks/types.py +308 -0
  377. fi/evals/metrics/code_security/detectors/__init__.py +186 -0
  378. fi/evals/metrics/code_security/detectors/base.py +394 -0
  379. fi/evals/metrics/code_security/detectors/cryptography.py +345 -0
  380. fi/evals/metrics/code_security/detectors/injection.py +744 -0
  381. fi/evals/metrics/code_security/detectors/secrets.py +287 -0
  382. fi/evals/metrics/code_security/detectors/serialization.py +192 -0
  383. fi/evals/metrics/code_security/joint_metrics.py +588 -0
  384. fi/evals/metrics/code_security/judges/__init__.py +83 -0
  385. fi/evals/metrics/code_security/judges/base.py +238 -0
  386. fi/evals/metrics/code_security/judges/dual_judge.py +534 -0
  387. fi/evals/metrics/code_security/judges/llm_judge.py +301 -0
  388. fi/evals/metrics/code_security/judges/pattern_judge.py +515 -0
  389. fi/evals/metrics/code_security/metrics.py +388 -0
  390. fi/evals/metrics/code_security/modes/__init__.py +63 -0
  391. fi/evals/metrics/code_security/modes/adversarial.py +284 -0
  392. fi/evals/metrics/code_security/modes/autocomplete.py +198 -0
  393. fi/evals/metrics/code_security/modes/base.py +283 -0
  394. fi/evals/metrics/code_security/modes/instruct.py +253 -0
  395. fi/evals/metrics/code_security/modes/repair.py +230 -0
  396. fi/evals/metrics/code_security/reports/__init__.py +57 -0
  397. fi/evals/metrics/code_security/reports/generator.py +404 -0
  398. fi/evals/metrics/code_security/reports/leaderboard.py +509 -0
  399. fi/evals/metrics/code_security/types.py +534 -0
  400. fi/evals/metrics/function_calling/__init__.py +34 -0
  401. fi/evals/metrics/function_calling/metrics.py +573 -0
  402. fi/evals/metrics/function_calling/types.py +87 -0
  403. fi/evals/metrics/hallucination/__init__.py +54 -0
  404. fi/evals/metrics/hallucination/detector.py +149 -0
  405. fi/evals/metrics/hallucination/metrics.py +390 -0
  406. fi/evals/metrics/hallucination/nli.py +253 -0
  407. fi/evals/metrics/hallucination/sentinel.py +106 -0
  408. fi/evals/metrics/hallucination/types.py +132 -0
  409. fi/evals/metrics/heuristics/aggregation_metrics.py +85 -0
  410. fi/evals/metrics/heuristics/json_metrics.py +87 -0
  411. fi/evals/metrics/heuristics/similarity_metrics.py +375 -0
  412. fi/evals/metrics/heuristics/string_metrics.py +391 -0
  413. fi/evals/metrics/llm_as_judges/__init__.py +17 -0
  414. fi/evals/metrics/llm_as_judges/custom_judge/metric.py +112 -0
  415. fi/evals/metrics/llm_as_judges/custom_judge/prompts.py +26 -0
  416. fi/evals/metrics/llm_as_judges/types.py +48 -0
  417. fi/evals/metrics/rag/__init__.py +111 -0
  418. fi/evals/metrics/rag/advanced/__init__.py +14 -0
  419. fi/evals/metrics/rag/advanced/multi_hop.py +283 -0
  420. fi/evals/metrics/rag/advanced/source_attribution.py +344 -0
  421. fi/evals/metrics/rag/generation/__init__.py +17 -0
  422. fi/evals/metrics/rag/generation/answer_relevancy.py +176 -0
  423. fi/evals/metrics/rag/generation/context_utilization.py +245 -0
  424. fi/evals/metrics/rag/generation/faithfulness.py +241 -0
  425. fi/evals/metrics/rag/generation/groundedness.py +131 -0
  426. fi/evals/metrics/rag/rag_score.py +277 -0
  427. fi/evals/metrics/rag/retrieval/__init__.py +20 -0
  428. fi/evals/metrics/rag/retrieval/context_entity_recall.py +124 -0
  429. fi/evals/metrics/rag/retrieval/context_precision.py +158 -0
  430. fi/evals/metrics/rag/retrieval/context_recall.py +106 -0
  431. fi/evals/metrics/rag/retrieval/noise_sensitivity.py +163 -0
  432. fi/evals/metrics/rag/retrieval/ranking.py +261 -0
  433. fi/evals/metrics/rag/types.py +100 -0
  434. fi/evals/metrics/rag/utils/__init__.py +62 -0
  435. fi/evals/metrics/rag/utils/claims.py +189 -0
  436. fi/evals/metrics/rag/utils/entities.py +244 -0
  437. fi/evals/metrics/rag/utils/nli.py +92 -0
  438. fi/evals/metrics/rag/utils/similarity.py +345 -0
  439. fi/evals/metrics/structured/__init__.py +114 -0
  440. fi/evals/metrics/structured/field_completeness.py +313 -0
  441. fi/evals/metrics/structured/hierarchy_score.py +366 -0
  442. fi/evals/metrics/structured/json_validation.py +190 -0
  443. fi/evals/metrics/structured/schema_compliance.py +280 -0
  444. fi/evals/metrics/structured/structured_output_score.py +298 -0
  445. fi/evals/metrics/structured/types.py +108 -0
  446. fi/evals/metrics/structured/validators/__init__.py +30 -0
  447. fi/evals/metrics/structured/validators/base.py +189 -0
  448. fi/evals/metrics/structured/validators/json_validator.py +196 -0
  449. fi/evals/metrics/structured/validators/pydantic_validator.py +178 -0
  450. fi/evals/metrics/structured/validators/yaml_validator.py +248 -0
  451. fi/evals/otel/__init__.py +266 -0
  452. fi/evals/otel/config.py +400 -0
  453. fi/evals/otel/conventions.py +463 -0
  454. fi/evals/otel/enrichment.py +371 -0
  455. fi/evals/otel/instrumentors/__init__.py +140 -0
  456. fi/evals/otel/instrumentors/anthropic.py +517 -0
  457. fi/evals/otel/instrumentors/base.py +382 -0
  458. fi/evals/otel/instrumentors/openai.py +673 -0
  459. fi/evals/otel/processors/__init__.py +36 -0
  460. fi/evals/otel/processors/base.py +473 -0
  461. fi/evals/otel/processors/cost.py +445 -0
  462. fi/evals/otel/processors/evaluation.py +559 -0
  463. fi/evals/otel/processors/llm.py +462 -0
  464. fi/evals/otel/tracer.py +506 -0
  465. fi/evals/otel/types.py +232 -0
  466. fi/evals/otel_utils.py +23 -0
  467. fi/evals/protect.py +671 -0
  468. fi/evals/protect_input_adapter.py +154 -0
  469. fi/evals/streaming/__init__.py +88 -0
  470. fi/evals/streaming/buffer.py +213 -0
  471. fi/evals/streaming/evaluator.py +551 -0
  472. fi/evals/streaming/policy.py +307 -0
  473. fi/evals/streaming/scorers.py +368 -0
  474. fi/evals/streaming/types.py +238 -0
  475. fi/evals/templates.py +472 -0
  476. fi/evals/types.py +156 -0
  477. fi/opt/__init__.py +221 -0
  478. fi/opt/_objective_scoring.py +85 -0
  479. fi/opt/base/__init__.py +11 -0
  480. fi/opt/base/base_generator.py +33 -0
  481. fi/opt/base/base_mapper.py +26 -0
  482. fi/opt/base/base_optimizer.py +45 -0
  483. fi/opt/base/evaluator.py +211 -0
  484. fi/opt/components.py +3095 -0
  485. fi/opt/datamappers/__init__.py +3 -0
  486. fi/opt/datamappers/basic_mapper.py +40 -0
  487. fi/opt/deployment.py +1021 -0
  488. fi/opt/evidence.py +4332 -0
  489. fi/opt/generators/__init__.py +3 -0
  490. fi/opt/generators/litellm.py +66 -0
  491. fi/opt/integrations/__init__.py +23 -0
  492. fi/opt/integrations/generative_suite.py +410 -0
  493. fi/opt/integrations/simulate.py +1313 -0
  494. fi/opt/mutations.py +771 -0
  495. fi/opt/observability.py +4639 -0
  496. fi/opt/optimizer_trace.py +889 -0
  497. fi/opt/optimizers/__init__.py +80 -0
  498. fi/opt/optimizers/agent.py +331 -0
  499. fi/opt/optimizers/agent_bandit.py +392 -0
  500. fi/opt/optimizers/agent_curriculum.py +635 -0
  501. fi/opt/optimizers/agent_evolution.py +894 -0
  502. fi/opt/optimizers/agent_feedback.py +1863 -0
  503. fi/opt/optimizers/agent_pareto.py +547 -0
  504. fi/opt/optimizers/agent_social_memory.py +1113 -0
  505. fi/opt/optimizers/agent_tpe.py +321 -0
  506. fi/opt/optimizers/bayesian_search.py +449 -0
  507. fi/opt/optimizers/council.py +2075 -0
  508. fi/opt/optimizers/futureagi_replay.py +799 -0
  509. fi/opt/optimizers/gepa.py +322 -0
  510. fi/opt/optimizers/metaprompt.py +243 -0
  511. fi/opt/optimizers/promptwizard.py +417 -0
  512. fi/opt/optimizers/protegi.py +329 -0
  513. fi/opt/optimizers/random_search.py +224 -0
  514. fi/opt/research.py +518 -0
  515. fi/opt/simulation.py +260 -0
  516. fi/opt/targets.py +232 -0
  517. fi/opt/types.py +66 -0
  518. fi/opt/utils/__init__.py +4 -0
  519. fi/opt/utils/early_stopping.py +266 -0
  520. fi/opt/utils/setup_logging.py +82 -0
  521. fi/simulate/__init__.py +540 -0
  522. fi/simulate/_hashing.py +35 -0
  523. fi/simulate/_logging.py +10 -0
  524. fi/simulate/adapters.py +87 -0
  525. fi/simulate/agent/__init__.py +120 -0
  526. fi/simulate/agent/browser.py +658 -0
  527. fi/simulate/agent/definition.py +587 -0
  528. fi/simulate/agent/frameworks.py +3528 -0
  529. fi/simulate/agent/generic.py +8286 -0
  530. fi/simulate/agent/import_probe.py +227 -0
  531. fi/simulate/agent/memory.py +905 -0
  532. fi/simulate/agent/mocks.py +101 -0
  533. fi/simulate/agent/multi_agent.py +361 -0
  534. fi/simulate/agent/orchestration.py +903 -0
  535. fi/simulate/agent/realtime.py +665 -0
  536. fi/simulate/agent/wrapper.py +99 -0
  537. fi/simulate/agent/wrappers/__init__.py +18 -0
  538. fi/simulate/agent/wrappers/anthropic.py +62 -0
  539. fi/simulate/agent/wrappers/gemini.py +65 -0
  540. fi/simulate/agent/wrappers/http.py +404 -0
  541. fi/simulate/agent/wrappers/langchain.py +80 -0
  542. fi/simulate/agent/wrappers/openai.py +75 -0
  543. fi/simulate/agent/wrappers/websocket.py +326 -0
  544. fi/simulate/artifacts/__init__.py +11 -0
  545. fi/simulate/artifacts/manifest.py +62 -0
  546. fi/simulate/cli.py +20560 -0
  547. fi/simulate/endpoints/__init__.py +45 -0
  548. fi/simulate/endpoints/_http_actor.py +73 -0
  549. fi/simulate/endpoints/actor_sources.py +243 -0
  550. fi/simulate/endpoints/base.py +107 -0
  551. fi/simulate/endpoints/builtins.py +10 -0
  552. fi/simulate/endpoints/callable.py +95 -0
  553. fi/simulate/endpoints/http.py +76 -0
  554. fi/simulate/endpoints/livekit.py +138 -0
  555. fi/simulate/endpoints/originators.py +132 -0
  556. fi/simulate/endpoints/profiles.py +348 -0
  557. fi/simulate/endpoints/retell.py +633 -0
  558. fi/simulate/endpoints/vapi.py +205 -0
  559. fi/simulate/endpoints/websocket.py +76 -0
  560. fi/simulate/environment.py +33026 -0
  561. fi/simulate/environments/__init__.py +11 -0
  562. fi/simulate/environments/base.py +73 -0
  563. fi/simulate/environments/chat.py +697 -0
  564. fi/simulate/environments/voice.py +212 -0
  565. fi/simulate/evaluation/__init__.py +4 -0
  566. fi/simulate/evaluation/ai_eval.py +227 -0
  567. fi/simulate/evidence/__init__.py +35 -0
  568. fi/simulate/evidence/base.py +59 -0
  569. fi/simulate/evidence/caller_observed.py +50 -0
  570. fi/simulate/evidence/livekit_instrumentation.py +51 -0
  571. fi/simulate/evidence/livekit_room.py +50 -0
  572. fi/simulate/evidence/otel.py +49 -0
  573. fi/simulate/evidence/providers/__init__.py +24 -0
  574. fi/simulate/evidence/providers/base.py +61 -0
  575. fi/simulate/evidence/providers/retell.py +376 -0
  576. fi/simulate/evidence/providers/vapi.py +426 -0
  577. fi/simulate/hosted/__init__.py +32 -0
  578. fi/simulate/hosted/child_entrypoint.py +306 -0
  579. fi/simulate/hosted/job.py +150 -0
  580. fi/simulate/hosted/targets.py +53 -0
  581. fi/simulate/instrumentation/__init__.py +5 -0
  582. fi/simulate/instrumentation/livekit/__init__.py +122 -0
  583. fi/simulate/manifest.py +1033 -0
  584. fi/simulate/matrix_cli.py +165 -0
  585. fi/simulate/realtime/__init__.py +40 -0
  586. fi/simulate/realtime/events.py +107 -0
  587. fi/simulate/realtime/media.py +61 -0
  588. fi/simulate/realtime/session.py +91 -0
  589. fi/simulate/recording/__init__.py +5 -0
  590. fi/simulate/recording/room_recorder.py +326 -0
  591. fi/simulate/registry.py +185 -0
  592. fi/simulate/results/__init__.py +9 -0
  593. fi/simulate/results/base.py +18 -0
  594. fi/simulate/results/filesystem.py +71 -0
  595. fi/simulate/results/futureagi.py +1340 -0
  596. fi/simulate/runtime/__init__.py +85 -0
  597. fi/simulate/runtime/capabilities.py +40 -0
  598. fi/simulate/runtime/events.py +63 -0
  599. fi/simulate/runtime/failures.py +25 -0
  600. fi/simulate/runtime/ids.py +34 -0
  601. fi/simulate/runtime/plan.py +70 -0
  602. fi/simulate/runtime/planner.py +102 -0
  603. fi/simulate/runtime/report.py +174 -0
  604. fi/simulate/runtime/run.py +75 -0
  605. fi/simulate/runtime/runner.py +333 -0
  606. fi/simulate/runtime/spec.py +186 -0
  607. fi/simulate/simulation/__init__.py +30 -0
  608. fi/simulate/simulation/behavior_policy.py +425 -0
  609. fi/simulate/simulation/bridge/__init__.py +9 -0
  610. fi/simulate/simulation/bridge/audio.py +29 -0
  611. fi/simulate/simulation/bridge/connector.py +46 -0
  612. fi/simulate/simulation/bridge/livekit.py +252 -0
  613. fi/simulate/simulation/bridge/retell.py +188 -0
  614. fi/simulate/simulation/bridge/vapi.py +177 -0
  615. fi/simulate/simulation/contract.py +419 -0
  616. fi/simulate/simulation/engines/__init__.py +12 -0
  617. fi/simulate/simulation/engines/base.py +21 -0
  618. fi/simulate/simulation/engines/cloud.py +517 -0
  619. fi/simulate/simulation/engines/livekit.py +4167 -0
  620. fi/simulate/simulation/engines/local_text.py +89 -0
  621. fi/simulate/simulation/fidelity.py +374 -0
  622. fi/simulate/simulation/gemini_tts_stream.py +110 -0
  623. fi/simulate/simulation/generator.py +91 -0
  624. fi/simulate/simulation/goal_machine.py +185 -0
  625. fi/simulate/simulation/livekit_models.py +467 -0
  626. fi/simulate/simulation/matrix.py +170 -0
  627. fi/simulate/simulation/models.py +279 -0
  628. fi/simulate/simulation/runner.py +153 -0
  629. fi/simulate/simulation/synthetic.py +880 -0
  630. fi/simulate/simulation/voice_prompt.py +502 -0
  631. fi/simulate/simulator/__init__.py +55 -0
  632. fi/simulate/simulator/builtins.py +53 -0
  633. fi/simulate/suite.py +1288 -0
  634. fi/simulate/utils/routes.py +164 -0
  635. fi/simulate/voice.py +225 -0
  636. fi/simulate/voice_cli.py +182 -0
  637. fi/utils/__init__.py +1 -0
  638. fi/utils/constants.py +14 -0
  639. fi/utils/errors.py +200 -0
  640. fi/utils/executor.py +26 -0
  641. fi/utils/routes.py +119 -0
  642. fi/utils/utils.py +17 -0
@@ -0,0 +1,515 @@
1
+ """Deterministic packaging admission for submitted agent repositories.
2
+
3
+ This pass never executes source or invents a runtime. It finds packaging the repository already
4
+ ships and catches common, expensive failures before Docker is started. The selected component is
5
+ also explicit, which matters for monorepositories containing several unrelated example agents.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import json
11
+ import re
12
+ import shlex
13
+ from enum import Enum
14
+ from pathlib import Path
15
+
16
+ from pydantic import BaseModel, Field
17
+ import yaml
18
+
19
+
20
+ class PackagingKind(str, Enum):
21
+ COMPOSE = "compose"
22
+ DOCKERFILE = "dockerfile"
23
+
24
+
25
+ class PackagingFinding(BaseModel):
26
+ code: str
27
+ message: str
28
+ blocking: bool = True
29
+
30
+
31
+ class PackagingCandidate(BaseModel):
32
+ path: str
33
+ kind: PackagingKind
34
+ findings: list[PackagingFinding] = Field(default_factory=list)
35
+ services: list[str] = Field(default_factory=list)
36
+ runtime_candidates: list[str] = Field(default_factory=list)
37
+ runtime_source_roots: list[str] = Field(default_factory=list)
38
+
39
+ @property
40
+ def viable(self) -> bool:
41
+ return not any(item.blocking for item in self.findings)
42
+
43
+
44
+ class PackagingManifest(BaseModel):
45
+ source_root: str
46
+ ready: bool
47
+ selected_path: str | None = None
48
+ selected_kind: PackagingKind | None = None
49
+ agent_runtime_packaged: bool = False
50
+ candidates: list[PackagingCandidate] = Field(default_factory=list)
51
+ notes: list[str] = Field(default_factory=list)
52
+
53
+
54
+ _IGNORED = {".git", ".venv", "node_modules", "vendor", "build", "dist", "artifacts"}
55
+ _COMPOSE_NAMES = {
56
+ "compose.yml",
57
+ "compose.yaml",
58
+ "docker-compose.yml",
59
+ "docker-compose.yaml",
60
+ }
61
+ _BIND_SOURCE = re.compile(r"--mount=type=bind,[^\n]*?source=([^,\s\\]+)")
62
+ _HOST_MOUNT = re.compile(
63
+ r"(?m)^\s*-\s*(?P<source>(?:/|~|\.|\$HOME|\$\{HOME\})[^:\n]*):(?P<target>/[^:\n]+)"
64
+ )
65
+
66
+
67
+ def inspect_packaging(
68
+ root: str | Path,
69
+ *,
70
+ max_depth: int = 4,
71
+ external_environment: bool = False,
72
+ ) -> PackagingManifest:
73
+ """Find and validate existing container packaging without running Docker."""
74
+ source = Path(root).expanduser().resolve()
75
+ if not source.is_dir():
76
+ raise ValueError(f"packaging_source_missing: {source}")
77
+
78
+ candidates: list[PackagingCandidate] = []
79
+ for path in sorted(source.rglob("*")):
80
+ relative = path.relative_to(source)
81
+ if len(relative.parts) > max_depth or any(
82
+ part in _IGNORED for part in relative.parts
83
+ ):
84
+ continue
85
+ if path.is_symlink() or not path.is_file():
86
+ continue
87
+ if _is_compose_file(path.name):
88
+ candidates.append(
89
+ _compose_candidate(
90
+ source,
91
+ path,
92
+ external_environment=external_environment,
93
+ )
94
+ )
95
+ elif _is_dockerfile(path.name):
96
+ candidates.append(_dockerfile_candidate(source, path))
97
+
98
+ viable = [item for item in candidates if item.viable]
99
+ selected: PackagingCandidate | None = None
100
+ root_compose = [
101
+ item
102
+ for item in viable
103
+ if item.kind is PackagingKind.COMPOSE and "/" not in item.path
104
+ ]
105
+ root_dockerfiles = [
106
+ item
107
+ for item in viable
108
+ if item.kind is PackagingKind.DOCKERFILE and "/" not in item.path
109
+ ]
110
+ development_compose = any(
111
+ finding.code == "compose_development_configuration"
112
+ for item in root_compose
113
+ for finding in item.findings
114
+ )
115
+ if len(root_compose) == 1 and not (
116
+ development_compose and len(root_dockerfiles) == 1
117
+ ):
118
+ selected = root_compose[0]
119
+ elif len(root_dockerfiles) == 1:
120
+ selected = root_dockerfiles[0]
121
+ elif len(viable) == 1:
122
+ selected = viable[0]
123
+
124
+ notes: list[str] = []
125
+ if not candidates:
126
+ # This is not inherently invalid: a remote-provider agent or a genuinely in-process
127
+ # agent may need no container runtime. The understanding stage decides that later.
128
+ notes.append(
129
+ "repository ships neither Compose nor a Dockerfile; runtime admission depends on "
130
+ "whether agent understanding finds external infrastructure"
131
+ )
132
+ elif not viable:
133
+ notes.append("all discovered packaging has blocking preflight findings")
134
+ elif selected is None:
135
+ notes.append(
136
+ "multiple runnable components were found; select the agent subdirectory or packaging path"
137
+ )
138
+ return PackagingManifest(
139
+ source_root=str(source),
140
+ ready=selected is not None or not candidates,
141
+ selected_path=selected.path if selected else None,
142
+ selected_kind=selected.kind if selected else None,
143
+ agent_runtime_packaged=(
144
+ selected is not None
145
+ and (
146
+ selected.kind is PackagingKind.DOCKERFILE
147
+ or bool(selected.runtime_candidates)
148
+ )
149
+ ),
150
+ candidates=candidates,
151
+ notes=notes,
152
+ )
153
+
154
+
155
+ def _is_dockerfile(name: str) -> bool:
156
+ """Accept Dockerfile variants without treating adjacent metadata as images."""
157
+ if name == "Dockerfile":
158
+ return True
159
+ if not name.startswith("Dockerfile."):
160
+ return False
161
+ return not name.endswith((".dockerignore", ".md", ".txt"))
162
+
163
+
164
+ def _is_compose_file(name: str) -> bool:
165
+ """Recognize standard and explicitly named Compose variants."""
166
+ lowered = name.lower()
167
+ if lowered in _COMPOSE_NAMES:
168
+ return True
169
+ if not lowered.endswith((".yml", ".yaml")):
170
+ return False
171
+ return lowered.startswith(("compose.", "docker-compose.")) and not any(
172
+ marker in lowered for marker in (".example.", ".sample.", ".bak.")
173
+ )
174
+
175
+
176
+ def _dockerfile_candidate(root: Path, path: Path) -> PackagingCandidate:
177
+ content = path.read_text(encoding="utf-8", errors="replace")
178
+ logical = content.replace("\\\n", " ")
179
+ findings: list[PackagingFinding] = []
180
+ if path.parent != root:
181
+ return PackagingCandidate(
182
+ path=path.relative_to(root).as_posix(),
183
+ kind=PackagingKind.DOCKERFILE,
184
+ findings=[
185
+ PackagingFinding(
186
+ code="dockerfile_context_selection_required",
187
+ message="Nested Dockerfile requires an explicit component root/build context",
188
+ blocking=False,
189
+ )
190
+ ],
191
+ )
192
+ sources = [match.group(1) for match in _BIND_SOURCE.finditer(logical)]
193
+ sources.extend(_dockerfile_copy_sources(logical))
194
+ for raw in sorted(set(sources)):
195
+ value = raw.strip("\"'")
196
+ if not value or value in {".", "./"}:
197
+ continue
198
+ if any(token in value for token in ("$", "*", "?", "[")):
199
+ continue
200
+ if value.startswith(("http://", "https://")):
201
+ continue
202
+ target = (root / value.removeprefix("./")).resolve()
203
+ try:
204
+ target.relative_to(root)
205
+ except ValueError:
206
+ findings.append(
207
+ PackagingFinding(
208
+ code="dockerfile_source_outside_repository",
209
+ message=f"Dockerfile references source outside the submitted root: {value}",
210
+ )
211
+ )
212
+ continue
213
+ if not target.exists():
214
+ findings.append(
215
+ PackagingFinding(
216
+ code="dockerfile_build_input_missing",
217
+ message=f"Dockerfile requires missing build input: {value}",
218
+ )
219
+ )
220
+ return PackagingCandidate(
221
+ path=path.relative_to(root).as_posix(),
222
+ kind=PackagingKind.DOCKERFILE,
223
+ findings=findings,
224
+ )
225
+
226
+
227
+ def _dockerfile_copy_sources(content: str) -> list[str]:
228
+ """Return every local source from shell- and JSON-form COPY/ADD instructions."""
229
+ sources: list[str] = []
230
+ for raw_line in content.splitlines():
231
+ match = re.match(r"^\s*(COPY|ADD)\s+(.+)$", raw_line, re.IGNORECASE)
232
+ if not match:
233
+ continue
234
+ arguments = match.group(2).strip()
235
+ if re.search(r"(?:^|\s)--from(?:=|\s)", arguments):
236
+ continue
237
+ while arguments.startswith("--"):
238
+ try:
239
+ option, arguments = arguments.split(None, 1)
240
+ except ValueError:
241
+ arguments = ""
242
+ break
243
+ # Options with a separate value consume that value as well.
244
+ if "=" not in option and option.lower() in {
245
+ "--chown",
246
+ "--chmod",
247
+ "--exclude",
248
+ }:
249
+ try:
250
+ _, arguments = arguments.split(None, 1)
251
+ except ValueError:
252
+ arguments = ""
253
+ break
254
+ if not arguments:
255
+ continue
256
+ if arguments.startswith("["):
257
+ try:
258
+ values = json.loads(arguments)
259
+ except json.JSONDecodeError:
260
+ continue
261
+ if isinstance(values, list) and len(values) >= 2:
262
+ sources.extend(str(value) for value in values[:-1])
263
+ continue
264
+ try:
265
+ values = shlex.split(arguments, comments=True)
266
+ except ValueError:
267
+ continue
268
+ if len(values) >= 2:
269
+ sources.extend(values[:-1])
270
+ return sources
271
+
272
+
273
+ def _compose_candidate(
274
+ root: Path,
275
+ path: Path,
276
+ *,
277
+ external_environment: bool = False,
278
+ ) -> PackagingCandidate:
279
+ content = path.read_text(encoding="utf-8", errors="replace")
280
+ findings: list[PackagingFinding] = []
281
+ services: dict[str, object] = {}
282
+ try:
283
+ document = yaml.safe_load(content) or {}
284
+ raw_services = (
285
+ document.get("services", {}) if isinstance(document, dict) else {}
286
+ )
287
+ if isinstance(raw_services, dict):
288
+ services = {str(name): value for name, value in raw_services.items()}
289
+ except yaml.YAMLError as exc:
290
+ findings.append(
291
+ PackagingFinding(
292
+ code="compose_yaml_invalid",
293
+ message=f"Compose YAML cannot be parsed: {exc}",
294
+ )
295
+ )
296
+ for service_name, raw_service in services.items():
297
+ if not isinstance(raw_service, dict):
298
+ continue
299
+ env_files = raw_service.get("env_file") or []
300
+ if isinstance(env_files, (str, dict)):
301
+ env_files = [env_files]
302
+ for raw_env_file in env_files:
303
+ optional = (
304
+ isinstance(raw_env_file, dict)
305
+ and raw_env_file.get("required") is False
306
+ )
307
+ value = (
308
+ str(raw_env_file.get("path") or "")
309
+ if isinstance(raw_env_file, dict)
310
+ else str(raw_env_file)
311
+ )
312
+ if not value or "$" in value:
313
+ continue
314
+ candidate = (path.parent / value).resolve()
315
+ try:
316
+ candidate.relative_to(root)
317
+ except ValueError:
318
+ findings.append(
319
+ PackagingFinding(
320
+ code="compose_env_file_outside_repository",
321
+ message=f"{service_name} reads an env file outside the repository: {value}",
322
+ )
323
+ )
324
+ continue
325
+ if not candidate.is_file() and not optional:
326
+ findings.append(
327
+ PackagingFinding(
328
+ code="compose_env_file_missing",
329
+ message=(
330
+ f"{service_name} uses uploaded environment values instead of "
331
+ f"repository secret file: {value}"
332
+ if external_environment
333
+ else f"{service_name} requires missing env file: {value}"
334
+ ),
335
+ blocking=not external_environment,
336
+ )
337
+ )
338
+ if services and all(
339
+ isinstance(service, dict)
340
+ and not service.get("image")
341
+ and not service.get("build")
342
+ for service in services.values()
343
+ ):
344
+ findings.append(
345
+ PackagingFinding(
346
+ code="compose_override_fragment",
347
+ message=(
348
+ "Compose file only overrides existing services and cannot run "
349
+ "as a standalone environment"
350
+ ),
351
+ )
352
+ )
353
+ for match in _HOST_MOUNT.finditer(content):
354
+ findings.append(
355
+ PackagingFinding(
356
+ code="compose_host_bind_mount",
357
+ message=f"Compose depends on host path {match.group('source')}",
358
+ # Local execution may deliberately use a repository-owned mount. A hosted
359
+ # provider must turn this finding into policy admission or mount a secret ref.
360
+ blocking=False,
361
+ )
362
+ )
363
+ development_signals = []
364
+ if re.search(r"(?m)^\s*(?:tty|stdin_open)\s*:\s*true\s*$", content, re.IGNORECASE):
365
+ development_signals.append("interactive terminal")
366
+ if re.search(r"(?m)^\s*container_name\s*:", content):
367
+ development_signals.append("fixed container name")
368
+ if re.search(r"(?m)^\s*-\s*['\"]?\d{2,5}-\d{2,5}:\d{2,5}-\d{2,5}", content):
369
+ development_signals.append("broad published port range")
370
+ if development_signals:
371
+ findings.append(
372
+ PackagingFinding(
373
+ code="compose_development_configuration",
374
+ message="Compose appears development-oriented: "
375
+ + ", ".join(development_signals),
376
+ blocking=False,
377
+ )
378
+ )
379
+ if re.search(r"(?m)^\s*privileged\s*:\s*true\s*$", content, re.IGNORECASE):
380
+ findings.append(
381
+ PackagingFinding(
382
+ code="compose_privileged",
383
+ message="Compose requests privileged container execution",
384
+ )
385
+ )
386
+ if re.search(
387
+ r"(?m)^\s*(?:network_mode|pid|ipc)\s*:\s*['\"]?host['\"]?\s*$", content
388
+ ):
389
+ findings.append(
390
+ PackagingFinding(
391
+ code="compose_host_namespace",
392
+ message="Compose requests a host namespace",
393
+ )
394
+ )
395
+ runtime_candidates = _compose_runtime_candidates(services)
396
+ return PackagingCandidate(
397
+ path=path.relative_to(root).as_posix(),
398
+ kind=PackagingKind.COMPOSE,
399
+ findings=findings,
400
+ services=sorted(services),
401
+ runtime_candidates=runtime_candidates,
402
+ runtime_source_roots=_compose_runtime_source_roots(
403
+ root, path, services, runtime_candidates
404
+ ),
405
+ )
406
+
407
+
408
+ def _compose_runtime_source_roots(
409
+ root: Path,
410
+ compose_path: Path,
411
+ services: dict[str, object],
412
+ runtime_candidates: list[str],
413
+ ) -> list[str]:
414
+ """Return submitted paths that can affect the selected application runtime."""
415
+ paths: set[str] = set()
416
+ for name in runtime_candidates:
417
+ raw_service = services.get(name)
418
+ service = raw_service if isinstance(raw_service, dict) else {}
419
+ build = service.get("build")
420
+ context = (
421
+ str(build.get("context") or ".")
422
+ if isinstance(build, dict)
423
+ else str(build or "")
424
+ )
425
+ if (
426
+ context
427
+ and "$" not in context
428
+ and not context.startswith(("http://", "https://"))
429
+ ):
430
+ candidate = (compose_path.parent / context).resolve()
431
+ try:
432
+ relative = candidate.relative_to(root)
433
+ except ValueError:
434
+ continue
435
+ if candidate.exists():
436
+ paths.add(relative.as_posix() or ".")
437
+ env_files = service.get("env_file") or []
438
+ if isinstance(env_files, (str, dict)):
439
+ env_files = [env_files]
440
+ for raw_env_file in env_files:
441
+ value = (
442
+ str(raw_env_file.get("path") or "")
443
+ if isinstance(raw_env_file, dict)
444
+ else str(raw_env_file)
445
+ )
446
+ if not value or "$" in value:
447
+ continue
448
+ candidate = (compose_path.parent / value).resolve()
449
+ try:
450
+ relative = candidate.relative_to(root)
451
+ except ValueError:
452
+ continue
453
+ if candidate.is_file():
454
+ paths.add(relative.as_posix())
455
+ return sorted(paths)
456
+
457
+
458
+ def _compose_runtime_candidates(services: dict[str, object]) -> list[str]:
459
+ """Identify application services without treating databases/admin UIs as agents."""
460
+ infrastructure = {
461
+ "postgres",
462
+ "postgresql",
463
+ "mysql",
464
+ "mariadb",
465
+ "redis",
466
+ "clickhouse",
467
+ "mongodb",
468
+ "mongo",
469
+ "rabbitmq",
470
+ "kafka",
471
+ "nats",
472
+ "minio",
473
+ "elasticsearch",
474
+ "opensearch",
475
+ "qdrant",
476
+ "neo4j",
477
+ }
478
+ administration = {"pgadmin", "redis-commander", "adminer", "grafana", "kibana"}
479
+ application_roles = {
480
+ "api",
481
+ "backend",
482
+ "server",
483
+ "voice-server",
484
+ "voice_server",
485
+ "orchestrator",
486
+ "runtime",
487
+ }
488
+ candidates: list[str] = []
489
+ for name, raw_service in services.items():
490
+ service = raw_service if isinstance(raw_service, dict) else {}
491
+ image = str(service.get("image") or "").lower()
492
+ haystack = f"{name} {image}".lower()
493
+ if name.lower() in administration or any(
494
+ word in haystack for word in administration
495
+ ):
496
+ continue
497
+ if name.lower() in application_roles:
498
+ candidates.append(name)
499
+ continue
500
+ if any(word in name.lower() for word in ("agent", "worker", "bot", "app")):
501
+ candidates.append(name)
502
+ continue
503
+ known_infrastructure = any(word in haystack for word in infrastructure)
504
+ if service.get("build") and not known_infrastructure:
505
+ candidates.append(name)
506
+ return sorted(candidates)
507
+
508
+
509
+ __all__ = [
510
+ "PackagingCandidate",
511
+ "PackagingFinding",
512
+ "PackagingKind",
513
+ "PackagingManifest",
514
+ "inspect_packaging",
515
+ ]
@@ -0,0 +1,157 @@
1
+ """The persona values the platform understands.
2
+
3
+ A persona field is only useful if the platform recognises what is in it: an accent it knows
4
+ selects a voice, a personality it knows attaches a sentence of behaviour guidance. A value
5
+ written in words of its own renders fine and then does nothing, which is how a suite ends up
6
+ with callers who all behave the same.
7
+
8
+ The values are read from the platform's own model when it is mounted, and from the copy carried
9
+ with the harness when it is not, so a writer is always offered real ones. The behaviour guidance
10
+ itself lives with the prompt builder, next to the code that applies it.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ import ast
16
+ import json
17
+ import logging
18
+ import os
19
+ from functools import lru_cache
20
+ from pathlib import Path
21
+
22
+ logger = logging.getLogger(__name__)
23
+
24
+ # Where the platform's tables are mounted. Colon-separated so voice and chat guides can both be
25
+ # offered; the first file defining a table wins, so voice takes precedence when both are present.
26
+ # Where the platform's persona model is mounted, for the values it accepts.
27
+ VOCABULARY_ENV = "HARNESS_PERSONA_VOCABULARY"
28
+
29
+ # The persona fields worth constraining, and the choice class each is drawn from. Only the ones
30
+ # that change behaviour or routing: a free-text occupation harms nothing, an accent nobody
31
+ # recognises silently loses the voice it was supposed to select.
32
+ FIELDS = {
33
+ "gender": "GenderChoices",
34
+ "age_group": "AgeGroupChoices",
35
+ "occupation": "ProfessionChoices",
36
+ "location": "LocationChoices",
37
+ "personality": "PersonalityChoices",
38
+ "communication_style": "CommunicationStyleChoices",
39
+ "accent": "AccentChoices",
40
+ "languages": "LanguageChoices",
41
+ }
42
+
43
+ # Constrained because something downstream reads them. The rest are offered as vocabulary but a
44
+ # writer who needs a value outside them is not stopped: an unknown occupation costs nothing,
45
+ # an unknown accent costs the voice.
46
+ ENFORCED = ("personality", "communication_style", "accent", "languages")
47
+
48
+
49
+ @lru_cache(maxsize=1)
50
+ def vocabulary() -> dict[str, list[str]]:
51
+ """What the platform accepts for each persona field.
52
+
53
+ Parsed out of the model's ``TextChoices`` classes for the same reason the guidance is read
54
+ rather than restated: the platform is the one that has to understand these values, so it is
55
+ the one that decides what they are. A persona written in words of its own renders fine, gets
56
+ no behaviour guidance, and cannot be grouped with anything on the platform afterwards.
57
+ """
58
+ path = os.environ.get(VOCABULARY_ENV) or ""
59
+ if not path or not Path(path).exists():
60
+ # No model mounted. Fall back to the copy carried with the harness so a writer is always
61
+ # offered real values: an empty vocabulary silently lets it invent an accent that selects
62
+ # no voice and a personality that attaches no guidance.
63
+ return _bundled_vocabulary()
64
+ try:
65
+ tree = ast.parse(Path(path).read_text(encoding="utf-8"))
66
+ except (OSError, SyntaxError):
67
+ logger.warning("persona vocabulary at %s is unreadable; using the bundled copy", path)
68
+ return _bundled_vocabulary()
69
+
70
+ by_class: dict[str, list[str]] = {}
71
+ for node in ast.walk(tree):
72
+ if not isinstance(node, ast.ClassDef):
73
+ continue
74
+ values: list[str] = []
75
+ for item in node.body:
76
+ if not isinstance(item, ast.Assign):
77
+ continue
78
+ try:
79
+ held = ast.literal_eval(item.value)
80
+ except ValueError:
81
+ continue
82
+ # ``NAME = "value", "Label"`` is the choices shape; a bare string is also accepted.
83
+ if isinstance(held, tuple) and held and isinstance(held[0], str):
84
+ values.append(held[0])
85
+ elif isinstance(held, str):
86
+ values.append(held)
87
+ if values:
88
+ by_class[node.name] = values
89
+
90
+ found = {
91
+ field: by_class[cls] for field, cls in FIELDS.items() if by_class.get(cls)
92
+ }
93
+ if not found:
94
+ # The file parsed but held none of the classes we key on, so it is the wrong file or the
95
+ # classes moved. Silently returning nothing would drop every persona constraint at once.
96
+ logger.warning(
97
+ "persona vocabulary at %s defines none of %s; using the bundled copy",
98
+ path,
99
+ ", ".join(sorted(set(FIELDS.values()))),
100
+ )
101
+ return _bundled_vocabulary()
102
+ return found
103
+
104
+
105
+
106
+ @lru_cache(maxsize=1)
107
+ def _bundled_vocabulary() -> dict[str, list[str]]:
108
+ """The platform's persona values, carried with the harness.
109
+
110
+ Kept so the harness constrains personas out of the box. Languages come from the agent
111
+ definition's set rather than the persona dropdown's two, because nothing on the platform
112
+ enforces the dropdown and a caller is expected to speak more than English and Hindi.
113
+ """
114
+ path = Path(__file__).parent / "data" / "persona_vocabulary.json"
115
+ try:
116
+ by_class = json.loads(path.read_text(encoding="utf-8"))
117
+ except (OSError, ValueError):
118
+ logger.warning("bundled persona vocabulary is unreadable; personas stay unconstrained")
119
+ return {}
120
+ return {
121
+ field: list(by_class[cls])
122
+ for field, cls in FIELDS.items()
123
+ if by_class.get(cls)
124
+ }
125
+
126
+
127
+ def offered(field: str) -> list[str]:
128
+ """The values this field accepts, or nothing if the platform's model was not readable."""
129
+ return list(vocabulary().get(field, []))
130
+
131
+
132
+ def unrecognised(persona: dict[str, object]) -> list[str]:
133
+ """Persona values the platform would not recognise, as sentences saying what to use instead.
134
+
135
+ Only the fields something downstream actually reads, and only when the vocabulary was found:
136
+ a harness that cannot see the platform's model must not start refusing personas over it.
137
+ """
138
+ known = vocabulary()
139
+ if not known:
140
+ return []
141
+ problems: list[str] = []
142
+ for field in ENFORCED:
143
+ allowed = known.get(field) or []
144
+ if not allowed:
145
+ continue
146
+ held = persona.get(field)
147
+ values = held if isinstance(held, list) else ([held] if held else [])
148
+ lowered = {str(one).strip().lower() for one in allowed}
149
+ for one in values:
150
+ text = str(one).strip()
151
+ if text and text.lower() not in lowered:
152
+ problems.append(
153
+ f"persona {field} {text!r} is not one the platform knows, so it will not "
154
+ f"reach the call. Use one of: {', '.join(allowed)}. Anything else this "
155
+ "person is like belongs in persona.metadata."
156
+ )
157
+ return problems