graph-agents-cli 0.3.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (291) hide show
  1. graph_agents_cli/__init__.py +26 -0
  2. graph_agents_cli/_api_policy.py +2145 -0
  3. graph_agents_cli/_approvals.py +400 -0
  4. graph_agents_cli/_build.py +186 -0
  5. graph_agents_cli/_build_info.json +7 -0
  6. graph_agents_cli/_chat_client.py +462 -0
  7. graph_agents_cli/_click.py +157 -0
  8. graph_agents_cli/_defaults.py +139 -0
  9. graph_agents_cli/_experiments.py +64 -0
  10. graph_agents_cli/_http.py +192 -0
  11. graph_agents_cli/_output.py +83 -0
  12. graph_agents_cli/_project.py +462 -0
  13. graph_agents_cli/_remote.py +220 -0
  14. graph_agents_cli/_response_schema.py +264 -0
  15. graph_agents_cli/_runner.py +319 -0
  16. graph_agents_cli/_skills_check.py +274 -0
  17. graph_agents_cli/_tools.py +189 -0
  18. graph_agents_cli/_trust.py +66 -0
  19. graph_agents_cli/api/__init__.py +15 -0
  20. graph_agents_cli/api/_changes.py +506 -0
  21. graph_agents_cli/api/_files.py +658 -0
  22. graph_agents_cli/api/cmd_api.py +2480 -0
  23. graph_agents_cli/deploy/__init__.py +15 -0
  24. graph_agents_cli/deploy/_config.py +171 -0
  25. graph_agents_cli/deploy/_image.py +128 -0
  26. graph_agents_cli/deploy/_kube.py +286 -0
  27. graph_agents_cli/deploy/_modes.py +234 -0
  28. graph_agents_cli/deploy/_preflight.py +370 -0
  29. graph_agents_cli/deploy/_values.py +168 -0
  30. graph_agents_cli/deploy/cmd_deploy.py +1866 -0
  31. graph_agents_cli/deploy/gitops.py +562 -0
  32. graph_agents_cli/deploy/local_load.py +273 -0
  33. graph_agents_cli/dev/__init__.py +13 -0
  34. graph_agents_cli/dev/cmd_build.py +131 -0
  35. graph_agents_cli/dev/cmd_install.py +78 -0
  36. graph_agents_cli/dev/cmd_lint.py +119 -0
  37. graph_agents_cli/dev/cmd_playground.py +297 -0
  38. graph_agents_cli/dev/policy_check.py +1287 -0
  39. graph_agents_cli/eval/__init__.py +22 -0
  40. graph_agents_cli/eval/_client.py +670 -0
  41. graph_agents_cli/eval/_common.py +177 -0
  42. graph_agents_cli/eval/_judge.py +168 -0
  43. graph_agents_cli/eval/_judge_runner.py +238 -0
  44. graph_agents_cli/eval/_paths.py +212 -0
  45. graph_agents_cli/eval/checks.py +581 -0
  46. graph_agents_cli/eval/cmd_analyze.py +278 -0
  47. graph_agents_cli/eval/cmd_compare.py +284 -0
  48. graph_agents_cli/eval/cmd_eval_group.py +80 -0
  49. graph_agents_cli/eval/cmd_generate.py +558 -0
  50. graph_agents_cli/eval/cmd_grade.py +466 -0
  51. graph_agents_cli/eval/cmd_metric.py +156 -0
  52. graph_agents_cli/eval/cmd_run.py +370 -0
  53. graph_agents_cli/eval/cmd_submit.py +400 -0
  54. graph_agents_cli/eval/config.py +435 -0
  55. graph_agents_cli/eval/dataset.py +350 -0
  56. graph_agents_cli/eval/gate.py +420 -0
  57. graph_agents_cli/eval/transcript.py +192 -0
  58. graph_agents_cli/extension/__init__.py +13 -0
  59. graph_agents_cli/extension/_compat.py +86 -0
  60. graph_agents_cli/extension/_loader.py +293 -0
  61. graph_agents_cli/extension/_manifest.py +135 -0
  62. graph_agents_cli/extension/_overrides.py +195 -0
  63. graph_agents_cli/extension/_paths.py +91 -0
  64. graph_agents_cli/extension/_refs.py +193 -0
  65. graph_agents_cli/extension/_resolver.py +453 -0
  66. graph_agents_cli/extension/_schema.py +106 -0
  67. graph_agents_cli/extension/_spec.py +253 -0
  68. graph_agents_cli/extension/_sync.py +102 -0
  69. graph_agents_cli/extension/_trust.py +58 -0
  70. graph_agents_cli/extension/cmd_extension_add.py +259 -0
  71. graph_agents_cli/extension/cmd_extension_group.py +57 -0
  72. graph_agents_cli/extension/cmd_extension_list.py +56 -0
  73. graph_agents_cli/extension/cmd_extension_remove.py +61 -0
  74. graph_agents_cli/extension/cmd_extension_update.py +195 -0
  75. graph_agents_cli/info/__init__.py +13 -0
  76. graph_agents_cli/info/cmd_info.py +222 -0
  77. graph_agents_cli/infra/__init__.py +15 -0
  78. graph_agents_cli/infra/checks.py +1169 -0
  79. graph_agents_cli/infra/cmd_infra.py +103 -0
  80. graph_agents_cli/main.py +591 -0
  81. graph_agents_cli/peer/__init__.py +15 -0
  82. graph_agents_cli/peer/_generate.py +254 -0
  83. graph_agents_cli/peer/cmd_peer.py +1151 -0
  84. graph_agents_cli/run/__init__.py +13 -0
  85. graph_agents_cli/run/_local_server.py +1157 -0
  86. graph_agents_cli/run/_signals.py +141 -0
  87. graph_agents_cli/run/cmd_approvals.py +530 -0
  88. graph_agents_cli/run/cmd_run.py +1421 -0
  89. graph_agents_cli/scaffold/__init__.py +19 -0
  90. graph_agents_cli/scaffold/agents/README.md +24 -0
  91. graph_agents_cli/scaffold/agents/empty_py/.template/templateconfig.yaml +22 -0
  92. graph_agents_cli/scaffold/agents/langgraph/.env.example +292 -0
  93. graph_agents_cli/scaffold/agents/langgraph/.template/templateconfig.yaml +28 -0
  94. graph_agents_cli/scaffold/agents/langgraph/Dockerfile +59 -0
  95. graph_agents_cli/scaffold/agents/langgraph/Dockerfile.langgraph-server +59 -0
  96. graph_agents_cli/scaffold/agents/langgraph/README.md +571 -0
  97. graph_agents_cli/scaffold/agents/langgraph/api-policy.yaml +60 -0
  98. graph_agents_cli/scaffold/agents/langgraph/app/__init__.py +20 -0
  99. graph_agents_cli/scaffold/agents/langgraph/app/agent.py +174 -0
  100. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/__init__.py +15 -0
  101. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/a2a.py +2162 -0
  102. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/a2a_client.py +1167 -0
  103. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/api_client.py +4220 -0
  104. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/approvals.py +1349 -0
  105. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/auth.py +1986 -0
  106. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/chat.py +2962 -0
  107. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/checkpointer.py +432 -0
  108. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/content.py +569 -0
  109. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/db.py +580 -0
  110. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/limits.py +203 -0
  111. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/metrics.py +231 -0
  112. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/middleware.py +361 -0
  113. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/model.py +611 -0
  114. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/playground.py +230 -0
  115. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/run_locks.py +459 -0
  116. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/structured.py +755 -0
  117. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/telemetry.py +681 -0
  118. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/threads.py +493 -0
  119. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/token_exchange.py +959 -0
  120. graph_agents_cli/scaffold/agents/langgraph/app/fast_api_app.py +770 -0
  121. graph_agents_cli/scaffold/agents/langgraph/app/policies/__init__.py +55 -0
  122. graph_agents_cli/scaffold/agents/langgraph/app/policies/custom.py +97 -0
  123. graph_agents_cli/scaffold/agents/langgraph/app/tools/__init__.py +46 -0
  124. graph_agents_cli/scaffold/agents/langgraph/app/tools/example_api.py +92 -0
  125. graph_agents_cli/scaffold/agents/langgraph/app/tools/weather.py +33 -0
  126. graph_agents_cli/scaffold/agents/langgraph/langgraph.json +14 -0
  127. graph_agents_cli/scaffold/agents/langgraph/pyproject.toml +78 -0
  128. graph_agents_cli/scaffold/agents/langgraph/tests/conftest.py +376 -0
  129. graph_agents_cli/scaffold/agents/langgraph/tests/eval/datasets/basic-dataset.json +53 -0
  130. graph_agents_cli/scaffold/agents/langgraph/tests/eval/eval_config.yaml +32 -0
  131. graph_agents_cli/scaffold/agents/langgraph/tests/integration/approval_graph.py +137 -0
  132. graph_agents_cli/scaffold/agents/langgraph/tests/integration/fake_issuer.py +216 -0
  133. graph_agents_cli/scaffold/agents/langgraph/tests/integration/fake_openai.py +357 -0
  134. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_a2a_outcomes.py +569 -0
  135. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_a2a_relay.py +479 -0
  136. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_api_surface.py +812 -0
  137. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_approvals.py +1367 -0
  138. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_approvals_server.py +794 -0
  139. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_cross_actor.py +497 -0
  140. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_cross_actor_server.py +247 -0
  141. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_history_repair.py +278 -0
  142. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_model_apis.py +242 -0
  143. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_postgres.py +637 -0
  144. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_resilience_postgres.py +770 -0
  145. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_runtime_guardrails.py +854 -0
  146. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_server_e2e.py +340 -0
  147. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_server_runtime.py +989 -0
  148. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_structured_answers.py +584 -0
  149. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_structured_server.py +222 -0
  150. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_token_exchange_issuer.py +650 -0
  151. graph_agents_cli/scaffold/agents/langgraph/tests/load_test/.results/.placeholder +0 -0
  152. graph_agents_cli/scaffold/agents/langgraph/tests/load_test/README.md +22 -0
  153. graph_agents_cli/scaffold/agents/langgraph/tests/load_test/conftest.py +21 -0
  154. graph_agents_cli/scaffold/agents/langgraph/tests/load_test/load_test.py +81 -0
  155. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_a2a_client.py +824 -0
  156. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_a2a_scoping.py +724 -0
  157. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_api_client.py +1214 -0
  158. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_api_client_hardening.py +716 -0
  159. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_api_policy_rpc.py +767 -0
  160. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_approval_ledger.py +1536 -0
  161. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_fake_model.py +115 -0
  162. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_jwt_policy.py +991 -0
  163. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_limits.py +310 -0
  164. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_logging.py +148 -0
  165. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_logging_hardening.py +271 -0
  166. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_policy.py +378 -0
  167. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_resilience.py +610 -0
  168. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_server_auth.py +702 -0
  169. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_structured.py +673 -0
  170. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_telemetry.py +404 -0
  171. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_thread_listing.py +255 -0
  172. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_threads.py +268 -0
  173. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_token_exchange.py +1320 -0
  174. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_untrusted_content.py +393 -0
  175. graph_agents_cli/scaffold/agents/langgraph/uv-fastapi.lock +2084 -0
  176. graph_agents_cli/scaffold/agents/langgraph/uv-langgraph-server.lock +2106 -0
  177. graph_agents_cli/scaffold/agents/langgraph/{{cookiecutter.agent_guidance_filename}} +129 -0
  178. graph_agents_cli/scaffold/base_templates/_shared/graph-agents-cli-manifest.yaml +36 -0
  179. graph_agents_cli/scaffold/base_templates/python/.dockerignore +32 -0
  180. graph_agents_cli/scaffold/base_templates/python/.github/CODEOWNERS +30 -0
  181. graph_agents_cli/scaffold/base_templates/python/.github/agent.env +7 -0
  182. graph_agents_cli/scaffold/base_templates/python/.github/workflows/pr_checks.yaml +214 -0
  183. graph_agents_cli/scaffold/base_templates/python/.gitignore +209 -0
  184. graph_agents_cli/scaffold/base_templates/python/tests/unit/test_dummy.py +23 -0
  185. graph_agents_cli/scaffold/base_templates/python/{{cookiecutter.agent_guidance_filename}} +35 -0
  186. graph_agents_cli/scaffold/cmd_scaffold_group.py +49 -0
  187. graph_agents_cli/scaffold/commands/__init__.py +13 -0
  188. graph_agents_cli/scaffold/commands/create.py +1424 -0
  189. graph_agents_cli/scaffold/commands/enhance.py +1652 -0
  190. graph_agents_cli/scaffold/commands/upgrade.py +570 -0
  191. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/.github/agent.env +12 -0
  192. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/.github/workflows/promote-to-prod.yaml +371 -0
  193. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/.github/workflows/staging.yaml +450 -0
  194. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/argocd/application-dev.yaml +43 -0
  195. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/argocd/application-prod.yaml +41 -0
  196. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/argocd/application-staging.yaml +43 -0
  197. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/.helmignore +14 -0
  198. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/Chart.yaml +21 -0
  199. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/examples/networkpolicy.yaml +103 -0
  200. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/NOTES.txt +48 -0
  201. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/_helpers.tpl +189 -0
  202. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/certificate.yaml +15 -0
  203. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/configmap.yaml +10 -0
  204. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/deployment.yaml +199 -0
  205. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/hpa.yaml +22 -0
  206. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/httproute.yaml +30 -0
  207. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/ingress.yaml +39 -0
  208. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/networkpolicy.yaml +48 -0
  209. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/pdb.yaml +13 -0
  210. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/postgresql-secret.yaml +37 -0
  211. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/service.yaml +15 -0
  212. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/serviceaccount.yaml +13 -0
  213. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/servicemonitor.yaml +42 -0
  214. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values-dev.yaml +22 -0
  215. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values-prod.yaml +45 -0
  216. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values-staging.yaml +29 -0
  217. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values.yaml +396 -0
  218. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/tests/integration/test_chart.py +269 -0
  219. graph_agents_cli/scaffold/deployment_targets/none/README.md +5 -0
  220. graph_agents_cli/scaffold/deployment_targets/none/python/README.md +6 -0
  221. graph_agents_cli/scaffold/utils/__init__.py +13 -0
  222. graph_agents_cli/scaffold/utils/backup.py +212 -0
  223. graph_agents_cli/scaffold/utils/build_record.py +257 -0
  224. graph_agents_cli/scaffold/utils/cli_options.py +184 -0
  225. graph_agents_cli/scaffold/utils/fs.py +83 -0
  226. graph_agents_cli/scaffold/utils/generate_locks.py +214 -0
  227. graph_agents_cli/scaffold/utils/generation_metadata.py +88 -0
  228. graph_agents_cli/scaffold/utils/keyedit.py +768 -0
  229. graph_agents_cli/scaffold/utils/keymerge.py +537 -0
  230. graph_agents_cli/scaffold/utils/language.py +138 -0
  231. graph_agents_cli/scaffold/utils/lock_utils.py +94 -0
  232. graph_agents_cli/scaffold/utils/logging.py +77 -0
  233. graph_agents_cli/scaffold/utils/manifest.py +292 -0
  234. graph_agents_cli/scaffold/utils/merge.py +970 -0
  235. graph_agents_cli/scaffold/utils/merge3.py +216 -0
  236. graph_agents_cli/scaffold/utils/openapi_seed.py +199 -0
  237. graph_agents_cli/scaffold/utils/remote_template.py +376 -0
  238. graph_agents_cli/scaffold/utils/template.py +1352 -0
  239. graph_agents_cli/scaffold/utils/upgrade.py +894 -0
  240. graph_agents_cli/scaffold/utils/version.py +438 -0
  241. graph_agents_cli/secrets/__init__.py +15 -0
  242. graph_agents_cli/secrets/_apply.py +954 -0
  243. graph_agents_cli/secrets/_required.py +188 -0
  244. graph_agents_cli/secrets/cmd_secrets.py +211 -0
  245. graph_agents_cli/setup/__init__.py +13 -0
  246. graph_agents_cli/setup/_antigravity.py +221 -0
  247. graph_agents_cli/setup/cmd_auth.py +1030 -0
  248. graph_agents_cli/setup/cmd_dev_token.py +513 -0
  249. graph_agents_cli/setup/cmd_setup.py +428 -0
  250. graph_agents_cli/setup/cmd_update.py +140 -0
  251. graph_agents_cli/skills/__init__.py +13 -0
  252. graph_agents_cli/skills/_bundle.py +65 -0
  253. graph_agents_cli/skills/data/README.md +19 -0
  254. graph_agents_cli/skills/data/graph-agents-cli-deploy/SKILL.md +357 -0
  255. graph_agents_cli/skills/data/graph-agents-cli-deploy/references/github-settings.md +113 -0
  256. graph_agents_cli/skills/data/graph-agents-cli-deploy/references/gitops.md +137 -0
  257. graph_agents_cli/skills/data/graph-agents-cli-deploy/references/kubernetes.md +315 -0
  258. graph_agents_cli/skills/data/graph-agents-cli-deploy/references/secrets.md +160 -0
  259. graph_agents_cli/skills/data/graph-agents-cli-eval/SKILL.md +303 -0
  260. graph_agents_cli/skills/data/graph-agents-cli-eval/references/dataset_schema.md +282 -0
  261. graph_agents_cli/skills/data/graph-agents-cli-eval/references/metrics-guide.md +143 -0
  262. graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/SKILL.md +659 -0
  263. graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/references/langchain-models.md +124 -0
  264. graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/references/langgraph.md +235 -0
  265. graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/references/template-contract.md +477 -0
  266. graph_agents_cli/skills/data/graph-agents-cli-observability/SKILL.md +231 -0
  267. graph_agents_cli/skills/data/graph-agents-cli-observability/references/langsmith.md +46 -0
  268. graph_agents_cli/skills/data/graph-agents-cli-observability/references/otel.md +59 -0
  269. graph_agents_cli/skills/data/graph-agents-cli-scaffold/SKILL.md +414 -0
  270. graph_agents_cli/skills/data/graph-agents-cli-scaffold/references/flags.md +134 -0
  271. graph_agents_cli/skills/data/graph-agents-cli-workflow/SKILL.md +478 -0
  272. graph_agents_cli/skills/data/graph-agents-cli-workflow/references/brainstorming.md +118 -0
  273. graph_agents_cli/skills/data/graph-agents-cli-workflow/references/commands.md +419 -0
  274. graph_agents_cli/skills/data/graph-agents-cli-workflow/references/extension.md +156 -0
  275. graph_agents_cli/skills/data/graph-agents-cli-workflow/references/internals.md +67 -0
  276. graph_agents_cli/skills/data/graph-agents-cli-workflow/references/spec-template.md +56 -0
  277. graph_agents_cli/skills/data/graph-agents-cli-workflow/references/terminology.md +119 -0
  278. graph_agents_cli/system/__init__.py +15 -0
  279. graph_agents_cli/system/_apply.py +519 -0
  280. graph_agents_cli/system/_checks.py +1023 -0
  281. graph_agents_cli/system/_deploy.py +215 -0
  282. graph_agents_cli/system/_model.py +363 -0
  283. graph_agents_cli/system/_system.py +664 -0
  284. graph_agents_cli/system/_views.py +208 -0
  285. graph_agents_cli/system/cmd_system.py +423 -0
  286. graph_agents_cli-0.3.1.dist-info/METADATA +162 -0
  287. graph_agents_cli-0.3.1.dist-info/RECORD +291 -0
  288. graph_agents_cli-0.3.1.dist-info/WHEEL +4 -0
  289. graph_agents_cli-0.3.1.dist-info/entry_points.txt +2 -0
  290. graph_agents_cli-0.3.1.dist-info/licenses/LICENSE +201 -0
  291. graph_agents_cli-0.3.1.dist-info/licenses/NOTICE +19 -0
@@ -0,0 +1,420 @@
1
+ # Copyright 2026 graph-agents-cli contributors
2
+ #
3
+ # Licensed under the Apache License, Version 2.0 (the "License");
4
+ # you may not use this file except in compliance with the License.
5
+ # You may obtain a copy of the License at
6
+ #
7
+ # https://www.apache.org/licenses/LICENSE-2.0
8
+ #
9
+ # Unless required by applicable law or agreed to in writing, software
10
+ # distributed under the License is distributed on an "AS IS" BASIS,
11
+ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
12
+ # See the License for the specific language governing permissions and
13
+ # limitations under the License.
14
+
15
+ """The evaluation gate: case statuses, quality rates, exit codes.
16
+
17
+ One rule: complete case accounting, every deterministic check and every
18
+ mandatory judge metric must pass. Only judge metrics listed under
19
+ ``quality_metrics:`` may miss their per-case threshold at a rate bounded by
20
+ ``min_pass_rate``. Statuses: ``passed``, ``failed``, ``quality_below_threshold``,
21
+ ``error``, ``missing``. Exit codes: 0 gate met; 1 any failed or a quality
22
+ metric under its rate; 2 any error or missing; 3 configuration error.
23
+ """
24
+
25
+ from __future__ import annotations
26
+
27
+ from dataclasses import dataclass, field
28
+ from typing import Any
29
+
30
+ from rich.markup import escape
31
+ from rich.table import Table
32
+
33
+ from graph_agents_cli._output import Console
34
+ from graph_agents_cli.eval._common import (
35
+ EXIT_GATE_FAILED,
36
+ EXIT_INCOMPLETE,
37
+ EXIT_OK,
38
+ EvalConfigError,
39
+ )
40
+ from graph_agents_cli.eval.checks import run_checks
41
+ from graph_agents_cli.eval.config import EvalConfig, render_judge_prompt, resolve_threshold
42
+ from graph_agents_cli.eval.dataset import SCOPE_ALL_TURNS, EvalCase
43
+ from graph_agents_cli.eval.transcript import RenderedCase, render_case
44
+
45
+ STATUS_PASSED = "passed"
46
+ STATUS_FAILED = "failed"
47
+ STATUS_QUALITY = "quality_below_threshold"
48
+ STATUS_ERROR = "error"
49
+ STATUS_MISSING = "missing"
50
+ STATUSES: tuple[str, ...] = (
51
+ STATUS_PASSED,
52
+ STATUS_FAILED,
53
+ STATUS_QUALITY,
54
+ STATUS_ERROR,
55
+ STATUS_MISSING,
56
+ )
57
+ # Higher is worse; used by ``eval compare`` to detect regressions.
58
+ STATUS_RANK: dict[str, int] = {
59
+ STATUS_PASSED: 0,
60
+ STATUS_QUALITY: 1,
61
+ STATUS_FAILED: 2,
62
+ STATUS_ERROR: 3,
63
+ STATUS_MISSING: 3,
64
+ }
65
+
66
+ _STATUS_STYLE = {
67
+ STATUS_PASSED: "green",
68
+ STATUS_QUALITY: "yellow",
69
+ STATUS_FAILED: "red",
70
+ STATUS_ERROR: "red",
71
+ STATUS_MISSING: "magenta",
72
+ }
73
+
74
+
75
+ @dataclass
76
+ class CaseGrade:
77
+ id: str
78
+ checks: dict[str, dict[str, Any]] = field(default_factory=dict)
79
+ judge_scores: dict[str, dict[str, Any]] = field(default_factory=dict)
80
+ missing: list[str] = field(default_factory=list)
81
+ errors: list[str] = field(default_factory=list)
82
+ failures: list[str] = field(default_factory=list)
83
+ quality_misses: list[str] = field(default_factory=list)
84
+ # What the judges of this case could not see in full (cut tool results,
85
+ # unrecorded earlier turns); reported, never silent.
86
+ judge_notes: list[str] = field(default_factory=list)
87
+ truncated_tool_results: int = 0
88
+ unrecorded_turns: int = 0
89
+ # `expect.scope: all_turns` on a multi-turn case whose trace has no per-turn
90
+ # records: its checks could read the final turn only.
91
+ checks_final_turn_only: bool = False
92
+
93
+ @property
94
+ def status(self) -> str:
95
+ if self.missing:
96
+ return STATUS_MISSING
97
+ if self.errors:
98
+ return STATUS_ERROR
99
+ if self.failures:
100
+ return STATUS_FAILED
101
+ if self.quality_misses:
102
+ return STATUS_QUALITY
103
+ return STATUS_PASSED
104
+
105
+ @property
106
+ def reasons(self) -> list[str]:
107
+ return [*self.missing, *self.errors, *self.failures, *self.quality_misses]
108
+
109
+ @property
110
+ def judgeable(self) -> bool:
111
+ """Judges run only for cases whose mandatory items have not already failed."""
112
+ return not (self.missing or self.errors or self.failures)
113
+
114
+ def to_dict(self) -> dict[str, Any]:
115
+ data: dict[str, Any] = {
116
+ "id": self.id,
117
+ "status": self.status,
118
+ "reasons": self.reasons,
119
+ "checks": self.checks,
120
+ "judge_scores": self.judge_scores,
121
+ }
122
+ if self.judge_notes:
123
+ data["judge_notes"] = list(self.judge_notes)
124
+ return data
125
+
126
+
127
+ # --- deterministic stage ------------------------------------------------------
128
+
129
+
130
+ def grade_deterministic(case: EvalCase, trace: dict[str, Any] | None) -> CaseGrade:
131
+ """Case accounting plus the deterministic checks the case declares."""
132
+ grade = CaseGrade(id=case.id)
133
+ if trace is None:
134
+ grade.missing.append("missing: no trace for case")
135
+ return grade
136
+ if trace.get("status") == "missing":
137
+ grade.missing.append(f"missing: {trace.get('error') or 'trace has no response'}")
138
+ return grade
139
+ if trace.get("status") == "error":
140
+ grade.errors.append(f"error: {trace.get('error') or 'generation failed'}")
141
+ return grade
142
+ if trace.get("response") is None:
143
+ grade.missing.append("missing: trace has no response")
144
+ return grade
145
+ grade.checks = run_checks(case.expect, trace)
146
+ turns = trace.get("turns")
147
+ grade.checks_final_turn_only = (
148
+ case.expect.get("scope") == SCOPE_ALL_TURNS
149
+ and len(case.user_messages()) > 1
150
+ and not (isinstance(turns, list) and turns)
151
+ )
152
+ for name, result in grade.checks.items():
153
+ if not result["passed"]:
154
+ grade.failures.append(f"{name}: {result['reason']}")
155
+ return grade
156
+
157
+
158
+ # --- judge stage --------------------------------------------------------------
159
+
160
+
161
+ def case_metrics(config: EvalConfig, case: EvalCase) -> dict[str, dict[str, Any]]:
162
+ """Metric name -> case-level spec for every judge/custom metric that applies."""
163
+ metrics: dict[str, dict[str, Any]] = {name: dict(spec) for name, spec in case.judge.items()}
164
+ for name in config.custom_metrics:
165
+ metrics.setdefault(name, {})
166
+ return metrics
167
+
168
+
169
+ def validate_case_metrics(config: EvalConfig, case: EvalCase) -> None:
170
+ """Unknown metric or missing threshold is a configuration error (exit 3)."""
171
+ known = config.metric_names()
172
+ for name, spec in case_metrics(config, case).items():
173
+ if name not in known:
174
+ raise EvalConfigError(
175
+ f"case {case.id!r} declares unknown judge metric {name!r} "
176
+ f"(known: {', '.join(sorted(known))})"
177
+ )
178
+ resolve_threshold(config, name, spec)
179
+
180
+
181
+ def item_id(case_id: str, metric: str) -> str:
182
+ return f"{case_id}/{metric}"
183
+
184
+
185
+ def plan_judge_items(
186
+ config: EvalConfig, case: EvalCase, trace: dict[str, Any], grade: CaseGrade
187
+ ) -> list[dict[str, Any]]:
188
+ """Runner items for the judge and custom metrics of one judgeable case."""
189
+ if not grade.judgeable:
190
+ return []
191
+ items: list[dict[str, Any]] = []
192
+ response = trace.get("response")
193
+ response = "" if response is None else str(response)
194
+ rendered: RenderedCase | None = None
195
+ for name, spec in case_metrics(config, case).items():
196
+ threshold = resolve_threshold(config, name, spec)
197
+ quality = config.is_quality(name)
198
+ custom = config.custom_metrics.get(name)
199
+ grade.judge_scores[name] = {
200
+ "score": None,
201
+ "threshold": threshold,
202
+ "passed": None,
203
+ "quality": quality,
204
+ "reasoning": "",
205
+ "error": None,
206
+ # "judge" (an LLM judge) or "custom" (a project callable): errors say which.
207
+ "kind": "custom" if custom is not None else "judge",
208
+ }
209
+ if custom is not None:
210
+ items.append(
211
+ {
212
+ "id": item_id(case.id, name),
213
+ "kind": "custom",
214
+ "case_id": case.id,
215
+ "metric": name,
216
+ "callable": custom.callable,
217
+ "case": case.raw,
218
+ "trace": trace,
219
+ }
220
+ )
221
+ continue
222
+ judge = config.judges[name]
223
+ if rendered is None:
224
+ # Judges see every turn (earlier replies and tool results), not only
225
+ # the user messages and the final reply.
226
+ rendered = render_case(case, trace, max_tool_result_chars=config.max_tool_result_chars)
227
+ grade.judge_notes = rendered.notes
228
+ grade.truncated_tool_results = rendered.truncated_results
229
+ grade.unrecorded_turns = rendered.unrecorded_turns
230
+ items.append(
231
+ {
232
+ "id": item_id(case.id, name),
233
+ "kind": "judge",
234
+ "case_id": case.id,
235
+ "metric": name,
236
+ "scale": judge.scale,
237
+ "prompt": render_judge_prompt(
238
+ judge,
239
+ conversation=rendered.conversation,
240
+ response=response,
241
+ reference=case.reference,
242
+ context=case.context,
243
+ tool_calls=rendered.final_tool_calls or None,
244
+ transcript=rendered.transcript,
245
+ max_tool_result_chars=config.max_tool_result_chars,
246
+ ),
247
+ }
248
+ )
249
+ return items
250
+
251
+
252
+ def apply_judge_results(grade: CaseGrade, results: dict[str, dict[str, Any]]) -> None:
253
+ """Fold runner results into the grade: error, mandatory fail, or quality miss."""
254
+ for name, entry in grade.judge_scores.items():
255
+ label = "custom metric" if entry.get("kind") == "custom" else "judge"
256
+ result = results.get(item_id(grade.id, name))
257
+ if result is None:
258
+ entry["error"] = "no result from judge runner"
259
+ grade.errors.append(f"error: {label} {name}: no result from judge runner")
260
+ continue
261
+ if result.get("error"):
262
+ entry["error"] = str(result["error"])
263
+ grade.errors.append(f"error: {label} {name}: {result['error']}")
264
+ continue
265
+ score = result.get("score")
266
+ if not isinstance(score, int | float) or isinstance(score, bool):
267
+ entry["error"] = f"{label} returned no numeric score ({score!r})"
268
+ grade.errors.append(f"error: {label} {name}: no numeric score")
269
+ continue
270
+ entry["score"] = float(score)
271
+ entry["reasoning"] = str(result.get("reasoning") or "")
272
+ passed = float(score) >= float(entry["threshold"])
273
+ entry["passed"] = passed
274
+ if passed:
275
+ continue
276
+ text = f"{name}: score {_fmt(score)} below threshold {_fmt(entry['threshold'])}"
277
+ if entry["quality"]:
278
+ grade.quality_misses.append(f"{text} (quality)")
279
+ else:
280
+ grade.failures.append(text)
281
+
282
+
283
+ def _fmt(value: Any) -> str:
284
+ if isinstance(value, float) and value.is_integer():
285
+ return str(int(value))
286
+ return str(value)
287
+
288
+
289
+ # --- aggregate ----------------------------------------------------------------
290
+
291
+
292
+ def summarize(grades: list[CaseGrade]) -> dict[str, int]:
293
+ summary = dict.fromkeys(STATUSES, 0)
294
+ for grade in grades:
295
+ summary[grade.status] += 1
296
+ return summary
297
+
298
+
299
+ def compute_quality(config: EvalConfig, grades: list[CaseGrade]) -> dict[str, dict[str, Any]]:
300
+ """Per quality metric: pass rate over the cases scored on it, and ``met``.
301
+
302
+ The denominator (``scored``) is the cases the metric actually scored: a
303
+ case that never declared the metric is not a pass. ``pass_rate`` and
304
+ ``met`` are null on an incomplete run (``status: incomplete``) and when no
305
+ case ran the metric (``status: not_run``, which cannot fail the gate: there
306
+ is nothing to measure, and every mandatory item still applies).
307
+ """
308
+ complete = bool(grades) and all(g.status not in (STATUS_ERROR, STATUS_MISSING) for g in grades)
309
+ quality: dict[str, dict[str, Any]] = {}
310
+ for name, metric in config.quality_metrics.items():
311
+ scored = [
312
+ entry
313
+ for g in grades
314
+ if (entry := g.judge_scores.get(name)) is not None
315
+ and entry.get("quality")
316
+ and entry.get("passed") is not None
317
+ ]
318
+ below = sum(1 for entry in scored if entry.get("passed") is False)
319
+ result: dict[str, Any] = {
320
+ "pass_rate": None,
321
+ "min_pass_rate": metric.min_pass_rate,
322
+ "met": None,
323
+ "below_threshold": below,
324
+ "scored": len(scored),
325
+ "passed": len(scored) - below,
326
+ }
327
+ if not complete:
328
+ result["status"] = "incomplete"
329
+ elif not scored:
330
+ result["status"] = "not_run"
331
+ else:
332
+ rate = (len(scored) - below) / len(scored)
333
+ result["pass_rate"] = round(rate, 4)
334
+ result["met"] = rate >= metric.min_pass_rate
335
+ result["status"] = "met" if result["met"] else "not_met"
336
+ quality[name] = result
337
+ return quality
338
+
339
+
340
+ def exit_code_for(summary: dict[str, int], quality: dict[str, dict[str, Any]]) -> int:
341
+ if summary.get(STATUS_ERROR) or summary.get(STATUS_MISSING):
342
+ return EXIT_INCOMPLETE
343
+ if summary.get(STATUS_FAILED) or any(q.get("met") is False for q in quality.values()):
344
+ return EXIT_GATE_FAILED
345
+ return EXIT_OK
346
+
347
+
348
+ def print_summary(console: Console, results: dict[str, Any]) -> None:
349
+ summary = results["summary"]
350
+ table = Table(title="Evaluation gate", show_header=True, header_style="bold")
351
+ table.add_column("Status")
352
+ table.add_column("Cases", justify="right")
353
+ for status in STATUSES:
354
+ count = summary.get(status, 0)
355
+ style = _STATUS_STYLE[status] if count else "dim"
356
+ table.add_row(f"[{style}]{status}[/{style}]", str(count))
357
+ table.add_row(
358
+ "[bold]planned[/bold]",
359
+ str(results.get("planned", sum(summary.get(s, 0) for s in STATUSES))),
360
+ )
361
+ console.print(table)
362
+
363
+ quality = results.get("quality") or {}
364
+ if quality:
365
+ qtable = Table(title="Quality metrics", show_header=True, header_style="bold")
366
+ qtable.add_column("Metric")
367
+ qtable.add_column("Pass rate", justify="right")
368
+ qtable.add_column("Passed/scored", justify="right")
369
+ qtable.add_column("Min", justify="right")
370
+ qtable.add_column("Met")
371
+ incomplete = bool(summary.get(STATUS_ERROR) or summary.get(STATUS_MISSING))
372
+ for name, entry in quality.items():
373
+ rate = entry.get("pass_rate")
374
+ met = entry.get("met")
375
+ if met is None:
376
+ not_run = entry.get("status") == "not_run" or (
377
+ not incomplete and entry.get("scored") == 0
378
+ )
379
+ met_text = "n/a (no case ran it)" if not_run else "n/a (incomplete)"
380
+ else:
381
+ met_text = "[green]yes[/green]" if met else "[red]no[/red]"
382
+ scored = entry.get("scored")
383
+ qtable.add_row(
384
+ name,
385
+ "n/a" if rate is None else f"{rate:.0%}",
386
+ "n/a" if scored is None else f"{entry.get('passed', 0)}/{scored}",
387
+ f"{entry.get('min_pass_rate', 1.0):.0%}",
388
+ met_text,
389
+ )
390
+ console.print(qtable)
391
+
392
+ flagged = [c for c in results.get("cases", []) if c["status"] != STATUS_PASSED]
393
+ if flagged:
394
+ ctable = Table(title="Cases needing attention", show_header=True, header_style="bold")
395
+ ctable.add_column("Case")
396
+ ctable.add_column("Status")
397
+ ctable.add_column("Reasons")
398
+ for case in flagged:
399
+ style = _STATUS_STYLE[case["status"]]
400
+ ctable.add_row(
401
+ case["id"],
402
+ f"[{style}]{case['status']}[/{style}]",
403
+ "\n".join(case["reasons"][:4]) + ("\n..." if len(case["reasons"]) > 4 else ""),
404
+ )
405
+ console.print(ctable)
406
+ warnings = results.get("warnings") or []
407
+ for text in warnings:
408
+ console.print(f"[bold yellow]Warning:[/bold yellow] [yellow]{escape(text)}[/yellow]")
409
+ code = summary.get("exit_code", 0)
410
+ verdict = {
411
+ 0: "[green]gate met[/green]",
412
+ 1: "[red]gate failed[/red]",
413
+ 2: "[magenta]incomplete run[/magenta]",
414
+ }
415
+ caveat = ""
416
+ if code == 0 and results.get("fake_model"):
417
+ caveat = (
418
+ " [bold yellow](fake model: plumbing check only, not a quality signal)[/bold yellow]"
419
+ )
420
+ console.print(f"Result: {verdict.get(code, code)} (exit code {code}){caveat}")
@@ -0,0 +1,192 @@
1
+ # Copyright 2026 graph-agents-cli contributors
2
+ #
3
+ # Licensed under the Apache License, Version 2.0 (the "License");
4
+ # you may not use this file except in compliance with the License.
5
+ # You may obtain a copy of the License at
6
+ #
7
+ # https://www.apache.org/licenses/LICENSE-2.0
8
+ #
9
+ # Unless required by applicable law or agreed to in writing, software
10
+ # distributed under the License is distributed on an "AS IS" BASIS,
11
+ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
12
+ # See the License for the specific language governing permissions and
13
+ # limitations under the License.
14
+
15
+ """What a judge sees of one case: the conversation, tool calls and replies of every turn.
16
+
17
+ A multi-turn case sends each user message on one thread, and the trace keeps
18
+ every turn under ``turns`` (the top-level ``response`` and ``tool_calls`` are
19
+ the final turn's). A judge scoring the final reply needs the earlier replies
20
+ and tool results: a follow-up that relies on an answer already given, or a
21
+ claim grounded in an earlier tool result, would otherwise look wrong.
22
+
23
+ Rendered pieces:
24
+
25
+ * ``conversation``: every turn before the reply being scored in full (user
26
+ message, each tool call with its result, the agent's reply), then the final
27
+ user message. Dataset ``system``/``assistant`` messages, which are not sent
28
+ to the agent, appear in place and are labelled as such.
29
+ * ``transcript``: the same plus the final turn's tool calls and reply.
30
+ * the final turn's tool calls, rendered by ``render_tool_calls``.
31
+
32
+ Tool results longer than the configured limit are cut with an explicit marker
33
+ that tells the judge how much it did not see, so a claim drawn from the omitted
34
+ part is never mistaken for an invented one.
35
+ """
36
+
37
+ from __future__ import annotations
38
+
39
+ import json
40
+ from dataclasses import dataclass, field
41
+ from typing import Any
42
+
43
+ from graph_agents_cli.eval.dataset import EvalCase
44
+
45
+ # Characters of one tool result shown to a judge before it is cut. The agent
46
+ # read the whole result, so the judge needs it too; the bound only keeps one
47
+ # runaway tool from exhausting the judge's context. Configurable per project as
48
+ # ``judge.max_tool_result_chars`` in eval_config.yaml (null = never cut).
49
+ DEFAULT_MAX_TOOL_RESULT_CHARS = 50_000
50
+
51
+ NOT_SENT = "(from the dataset; not sent to the agent)"
52
+
53
+
54
+ @dataclass
55
+ class RenderedCase:
56
+ conversation: str
57
+ transcript: str
58
+ final_tool_calls: list[dict[str, Any]] = field(default_factory=list)
59
+ # Tool results (in any turn) cut for the judge, and how many characters went unseen.
60
+ truncated_results: int = 0
61
+ truncated_chars: int = 0
62
+ # Earlier turns whose replies the trace does not record (an older trace or a
63
+ # generate override): the judge is told, rather than shown a gap silently.
64
+ unrecorded_turns: int = 0
65
+
66
+ @property
67
+ def notes(self) -> list[str]:
68
+ notes: list[str] = []
69
+ if self.truncated_results:
70
+ notes.append(
71
+ f"{self.truncated_results} tool result(s) cut for the judge "
72
+ f"({self.truncated_chars} characters not shown; raise "
73
+ "judge.max_tool_result_chars in eval_config.yaml to show more)"
74
+ )
75
+ if self.unrecorded_turns:
76
+ notes.append(
77
+ f"{self.unrecorded_turns} earlier turn(s) have no recorded reply in the trace "
78
+ "(re-run eval generate to record every turn)"
79
+ )
80
+ return notes
81
+
82
+
83
+ def _as_text(value: Any) -> str:
84
+ if value is None:
85
+ return ""
86
+ if isinstance(value, str):
87
+ return value
88
+ try:
89
+ return json.dumps(value, ensure_ascii=False, sort_keys=True, default=str)
90
+ except (TypeError, ValueError):
91
+ return str(value)
92
+
93
+
94
+ def _args_text(args: Any) -> str:
95
+ if args is None or args == {}:
96
+ return ""
97
+ return _as_text(args)
98
+
99
+
100
+ def cut_tool_result(text: str, max_chars: int | None) -> tuple[str, int]:
101
+ """``(text shown to the judge, characters cut)``; the cut is marked in-band."""
102
+ if max_chars is None or len(text) <= max_chars:
103
+ return text, 0
104
+ omitted = len(text) - max_chars
105
+ marker = (
106
+ f" [TRUNCATED by graph-agents-cli: the judge sees the first {max_chars} of "
107
+ f"{len(text)} characters of this tool result; {omitted} more were returned to the "
108
+ "agent but are not shown here. Do not treat a claim as unsupported only because "
109
+ "it could come from the omitted part.]"
110
+ )
111
+ return text[:max_chars] + marker, omitted
112
+
113
+
114
+ def render_tool_call(call: dict[str, Any], max_chars: int | None) -> tuple[str, int]:
115
+ """``name(args) -> result`` for one call, and how many result characters were cut."""
116
+ result, omitted = cut_tool_result(_as_text(call.get("result")), max_chars)
117
+ flag = " (error)" if call.get("is_error") else ""
118
+ return f"{call.get('name')}({_args_text(call.get('args'))}) -> {result}{flag}", omitted
119
+
120
+
121
+ def render_tool_calls(
122
+ calls: list[dict[str, Any]] | None, max_chars: int | None
123
+ ) -> tuple[list[str], int, int]:
124
+ """Rendered lines, the number of results cut, and the characters cut."""
125
+ lines: list[str] = []
126
+ cut = 0
127
+ omitted_total = 0
128
+ for call in calls or []:
129
+ if not isinstance(call, dict):
130
+ continue
131
+ line, omitted = render_tool_call(call, max_chars)
132
+ lines.append(line)
133
+ if omitted:
134
+ cut += 1
135
+ omitted_total += omitted
136
+ return lines, cut, omitted_total
137
+
138
+
139
+ def render_case(
140
+ case: EvalCase,
141
+ trace: dict[str, Any],
142
+ *,
143
+ max_tool_result_chars: int | None = DEFAULT_MAX_TOOL_RESULT_CHARS,
144
+ ) -> RenderedCase:
145
+ """Render the conversation and full transcript of one graded case."""
146
+ user_count = sum(1 for m in case.messages if m.get("role") == "user")
147
+ raw_turns = trace.get("turns")
148
+ turns: list[dict[str, Any]] = (
149
+ [t if isinstance(t, dict) else {} for t in raw_turns] if isinstance(raw_turns, list) else []
150
+ )
151
+ final_calls = [c for c in (trace.get("tool_calls") or []) if isinstance(c, dict)]
152
+ final_response = trace.get("response")
153
+ labelled = user_count > 1
154
+
155
+ before: list[str] = [] # everything up to and including the final user message
156
+ after: list[str] = [] # the final turn's tool calls and reply (transcript only)
157
+ rendered = RenderedCase(conversation="", transcript="", final_tool_calls=final_calls)
158
+
159
+ def _calls(lines_out: list[str], calls: list[dict[str, Any]]) -> None:
160
+ lines, cut, omitted = render_tool_calls(calls, max_tool_result_chars)
161
+ rendered.truncated_results += cut
162
+ rendered.truncated_chars += omitted
163
+ lines_out.extend(f"agent tool call: {line}" for line in lines)
164
+
165
+ turn_index = -1
166
+ for message in case.messages:
167
+ role = message.get("role", "?")
168
+ content = str(message.get("content", ""))
169
+ if role != "user":
170
+ # Dataset context for the judge; the agent never received it.
171
+ before.append(f"{role} {NOT_SENT}: {content}")
172
+ continue
173
+ turn_index += 1
174
+ if labelled:
175
+ before.append(f"--- turn {turn_index + 1} of {user_count} ---")
176
+ before.append(f"user: {content}")
177
+ if turn_index == user_count - 1:
178
+ continue # the final turn's calls and reply are the ones being scored
179
+ record = turns[turn_index] if turn_index < len(turns) else None
180
+ if record is None:
181
+ rendered.unrecorded_turns += 1
182
+ before.append("agent: [reply not recorded in the trace]")
183
+ continue
184
+ _calls(before, [c for c in (record.get("tool_calls") or []) if isinstance(c, dict)])
185
+ before.append(f"agent: {_as_text(record.get('response'))}")
186
+
187
+ _calls(after, final_calls)
188
+ after.append(f"agent: {_as_text(final_response)}")
189
+
190
+ rendered.conversation = "\n".join(before)
191
+ rendered.transcript = "\n".join(before + after)
192
+ return rendered
@@ -0,0 +1,13 @@
1
+ # Copyright 2026 Google LLC
2
+ #
3
+ # Licensed under the Apache License, Version 2.0 (the "License");
4
+ # you may not use this file except in compliance with the License.
5
+ # You may obtain a copy of the License at
6
+ #
7
+ # https://www.apache.org/licenses/LICENSE-2.0
8
+ #
9
+ # Unless required by applicable law or agreed to in writing, software
10
+ # distributed under the License is distributed on an "AS IS" BASIS,
11
+ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
12
+ # See the License for the specific language governing permissions and
13
+ # limitations under the License.