graph-agents-cli 0.3.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (291) hide show
  1. graph_agents_cli/__init__.py +26 -0
  2. graph_agents_cli/_api_policy.py +2145 -0
  3. graph_agents_cli/_approvals.py +400 -0
  4. graph_agents_cli/_build.py +186 -0
  5. graph_agents_cli/_build_info.json +7 -0
  6. graph_agents_cli/_chat_client.py +462 -0
  7. graph_agents_cli/_click.py +157 -0
  8. graph_agents_cli/_defaults.py +139 -0
  9. graph_agents_cli/_experiments.py +64 -0
  10. graph_agents_cli/_http.py +192 -0
  11. graph_agents_cli/_output.py +83 -0
  12. graph_agents_cli/_project.py +462 -0
  13. graph_agents_cli/_remote.py +220 -0
  14. graph_agents_cli/_response_schema.py +264 -0
  15. graph_agents_cli/_runner.py +319 -0
  16. graph_agents_cli/_skills_check.py +274 -0
  17. graph_agents_cli/_tools.py +189 -0
  18. graph_agents_cli/_trust.py +66 -0
  19. graph_agents_cli/api/__init__.py +15 -0
  20. graph_agents_cli/api/_changes.py +506 -0
  21. graph_agents_cli/api/_files.py +658 -0
  22. graph_agents_cli/api/cmd_api.py +2480 -0
  23. graph_agents_cli/deploy/__init__.py +15 -0
  24. graph_agents_cli/deploy/_config.py +171 -0
  25. graph_agents_cli/deploy/_image.py +128 -0
  26. graph_agents_cli/deploy/_kube.py +286 -0
  27. graph_agents_cli/deploy/_modes.py +234 -0
  28. graph_agents_cli/deploy/_preflight.py +370 -0
  29. graph_agents_cli/deploy/_values.py +168 -0
  30. graph_agents_cli/deploy/cmd_deploy.py +1866 -0
  31. graph_agents_cli/deploy/gitops.py +562 -0
  32. graph_agents_cli/deploy/local_load.py +273 -0
  33. graph_agents_cli/dev/__init__.py +13 -0
  34. graph_agents_cli/dev/cmd_build.py +131 -0
  35. graph_agents_cli/dev/cmd_install.py +78 -0
  36. graph_agents_cli/dev/cmd_lint.py +119 -0
  37. graph_agents_cli/dev/cmd_playground.py +297 -0
  38. graph_agents_cli/dev/policy_check.py +1287 -0
  39. graph_agents_cli/eval/__init__.py +22 -0
  40. graph_agents_cli/eval/_client.py +670 -0
  41. graph_agents_cli/eval/_common.py +177 -0
  42. graph_agents_cli/eval/_judge.py +168 -0
  43. graph_agents_cli/eval/_judge_runner.py +238 -0
  44. graph_agents_cli/eval/_paths.py +212 -0
  45. graph_agents_cli/eval/checks.py +581 -0
  46. graph_agents_cli/eval/cmd_analyze.py +278 -0
  47. graph_agents_cli/eval/cmd_compare.py +284 -0
  48. graph_agents_cli/eval/cmd_eval_group.py +80 -0
  49. graph_agents_cli/eval/cmd_generate.py +558 -0
  50. graph_agents_cli/eval/cmd_grade.py +466 -0
  51. graph_agents_cli/eval/cmd_metric.py +156 -0
  52. graph_agents_cli/eval/cmd_run.py +370 -0
  53. graph_agents_cli/eval/cmd_submit.py +400 -0
  54. graph_agents_cli/eval/config.py +435 -0
  55. graph_agents_cli/eval/dataset.py +350 -0
  56. graph_agents_cli/eval/gate.py +420 -0
  57. graph_agents_cli/eval/transcript.py +192 -0
  58. graph_agents_cli/extension/__init__.py +13 -0
  59. graph_agents_cli/extension/_compat.py +86 -0
  60. graph_agents_cli/extension/_loader.py +293 -0
  61. graph_agents_cli/extension/_manifest.py +135 -0
  62. graph_agents_cli/extension/_overrides.py +195 -0
  63. graph_agents_cli/extension/_paths.py +91 -0
  64. graph_agents_cli/extension/_refs.py +193 -0
  65. graph_agents_cli/extension/_resolver.py +453 -0
  66. graph_agents_cli/extension/_schema.py +106 -0
  67. graph_agents_cli/extension/_spec.py +253 -0
  68. graph_agents_cli/extension/_sync.py +102 -0
  69. graph_agents_cli/extension/_trust.py +58 -0
  70. graph_agents_cli/extension/cmd_extension_add.py +259 -0
  71. graph_agents_cli/extension/cmd_extension_group.py +57 -0
  72. graph_agents_cli/extension/cmd_extension_list.py +56 -0
  73. graph_agents_cli/extension/cmd_extension_remove.py +61 -0
  74. graph_agents_cli/extension/cmd_extension_update.py +195 -0
  75. graph_agents_cli/info/__init__.py +13 -0
  76. graph_agents_cli/info/cmd_info.py +222 -0
  77. graph_agents_cli/infra/__init__.py +15 -0
  78. graph_agents_cli/infra/checks.py +1169 -0
  79. graph_agents_cli/infra/cmd_infra.py +103 -0
  80. graph_agents_cli/main.py +591 -0
  81. graph_agents_cli/peer/__init__.py +15 -0
  82. graph_agents_cli/peer/_generate.py +254 -0
  83. graph_agents_cli/peer/cmd_peer.py +1151 -0
  84. graph_agents_cli/run/__init__.py +13 -0
  85. graph_agents_cli/run/_local_server.py +1157 -0
  86. graph_agents_cli/run/_signals.py +141 -0
  87. graph_agents_cli/run/cmd_approvals.py +530 -0
  88. graph_agents_cli/run/cmd_run.py +1421 -0
  89. graph_agents_cli/scaffold/__init__.py +19 -0
  90. graph_agents_cli/scaffold/agents/README.md +24 -0
  91. graph_agents_cli/scaffold/agents/empty_py/.template/templateconfig.yaml +22 -0
  92. graph_agents_cli/scaffold/agents/langgraph/.env.example +292 -0
  93. graph_agents_cli/scaffold/agents/langgraph/.template/templateconfig.yaml +28 -0
  94. graph_agents_cli/scaffold/agents/langgraph/Dockerfile +59 -0
  95. graph_agents_cli/scaffold/agents/langgraph/Dockerfile.langgraph-server +59 -0
  96. graph_agents_cli/scaffold/agents/langgraph/README.md +571 -0
  97. graph_agents_cli/scaffold/agents/langgraph/api-policy.yaml +60 -0
  98. graph_agents_cli/scaffold/agents/langgraph/app/__init__.py +20 -0
  99. graph_agents_cli/scaffold/agents/langgraph/app/agent.py +174 -0
  100. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/__init__.py +15 -0
  101. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/a2a.py +2162 -0
  102. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/a2a_client.py +1167 -0
  103. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/api_client.py +4220 -0
  104. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/approvals.py +1349 -0
  105. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/auth.py +1986 -0
  106. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/chat.py +2962 -0
  107. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/checkpointer.py +432 -0
  108. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/content.py +569 -0
  109. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/db.py +580 -0
  110. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/limits.py +203 -0
  111. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/metrics.py +231 -0
  112. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/middleware.py +361 -0
  113. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/model.py +611 -0
  114. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/playground.py +230 -0
  115. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/run_locks.py +459 -0
  116. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/structured.py +755 -0
  117. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/telemetry.py +681 -0
  118. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/threads.py +493 -0
  119. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/token_exchange.py +959 -0
  120. graph_agents_cli/scaffold/agents/langgraph/app/fast_api_app.py +770 -0
  121. graph_agents_cli/scaffold/agents/langgraph/app/policies/__init__.py +55 -0
  122. graph_agents_cli/scaffold/agents/langgraph/app/policies/custom.py +97 -0
  123. graph_agents_cli/scaffold/agents/langgraph/app/tools/__init__.py +46 -0
  124. graph_agents_cli/scaffold/agents/langgraph/app/tools/example_api.py +92 -0
  125. graph_agents_cli/scaffold/agents/langgraph/app/tools/weather.py +33 -0
  126. graph_agents_cli/scaffold/agents/langgraph/langgraph.json +14 -0
  127. graph_agents_cli/scaffold/agents/langgraph/pyproject.toml +78 -0
  128. graph_agents_cli/scaffold/agents/langgraph/tests/conftest.py +376 -0
  129. graph_agents_cli/scaffold/agents/langgraph/tests/eval/datasets/basic-dataset.json +53 -0
  130. graph_agents_cli/scaffold/agents/langgraph/tests/eval/eval_config.yaml +32 -0
  131. graph_agents_cli/scaffold/agents/langgraph/tests/integration/approval_graph.py +137 -0
  132. graph_agents_cli/scaffold/agents/langgraph/tests/integration/fake_issuer.py +216 -0
  133. graph_agents_cli/scaffold/agents/langgraph/tests/integration/fake_openai.py +357 -0
  134. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_a2a_outcomes.py +569 -0
  135. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_a2a_relay.py +479 -0
  136. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_api_surface.py +812 -0
  137. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_approvals.py +1367 -0
  138. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_approvals_server.py +794 -0
  139. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_cross_actor.py +497 -0
  140. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_cross_actor_server.py +247 -0
  141. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_history_repair.py +278 -0
  142. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_model_apis.py +242 -0
  143. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_postgres.py +637 -0
  144. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_resilience_postgres.py +770 -0
  145. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_runtime_guardrails.py +854 -0
  146. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_server_e2e.py +340 -0
  147. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_server_runtime.py +989 -0
  148. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_structured_answers.py +584 -0
  149. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_structured_server.py +222 -0
  150. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_token_exchange_issuer.py +650 -0
  151. graph_agents_cli/scaffold/agents/langgraph/tests/load_test/.results/.placeholder +0 -0
  152. graph_agents_cli/scaffold/agents/langgraph/tests/load_test/README.md +22 -0
  153. graph_agents_cli/scaffold/agents/langgraph/tests/load_test/conftest.py +21 -0
  154. graph_agents_cli/scaffold/agents/langgraph/tests/load_test/load_test.py +81 -0
  155. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_a2a_client.py +824 -0
  156. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_a2a_scoping.py +724 -0
  157. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_api_client.py +1214 -0
  158. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_api_client_hardening.py +716 -0
  159. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_api_policy_rpc.py +767 -0
  160. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_approval_ledger.py +1536 -0
  161. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_fake_model.py +115 -0
  162. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_jwt_policy.py +991 -0
  163. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_limits.py +310 -0
  164. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_logging.py +148 -0
  165. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_logging_hardening.py +271 -0
  166. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_policy.py +378 -0
  167. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_resilience.py +610 -0
  168. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_server_auth.py +702 -0
  169. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_structured.py +673 -0
  170. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_telemetry.py +404 -0
  171. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_thread_listing.py +255 -0
  172. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_threads.py +268 -0
  173. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_token_exchange.py +1320 -0
  174. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_untrusted_content.py +393 -0
  175. graph_agents_cli/scaffold/agents/langgraph/uv-fastapi.lock +2084 -0
  176. graph_agents_cli/scaffold/agents/langgraph/uv-langgraph-server.lock +2106 -0
  177. graph_agents_cli/scaffold/agents/langgraph/{{cookiecutter.agent_guidance_filename}} +129 -0
  178. graph_agents_cli/scaffold/base_templates/_shared/graph-agents-cli-manifest.yaml +36 -0
  179. graph_agents_cli/scaffold/base_templates/python/.dockerignore +32 -0
  180. graph_agents_cli/scaffold/base_templates/python/.github/CODEOWNERS +30 -0
  181. graph_agents_cli/scaffold/base_templates/python/.github/agent.env +7 -0
  182. graph_agents_cli/scaffold/base_templates/python/.github/workflows/pr_checks.yaml +214 -0
  183. graph_agents_cli/scaffold/base_templates/python/.gitignore +209 -0
  184. graph_agents_cli/scaffold/base_templates/python/tests/unit/test_dummy.py +23 -0
  185. graph_agents_cli/scaffold/base_templates/python/{{cookiecutter.agent_guidance_filename}} +35 -0
  186. graph_agents_cli/scaffold/cmd_scaffold_group.py +49 -0
  187. graph_agents_cli/scaffold/commands/__init__.py +13 -0
  188. graph_agents_cli/scaffold/commands/create.py +1424 -0
  189. graph_agents_cli/scaffold/commands/enhance.py +1652 -0
  190. graph_agents_cli/scaffold/commands/upgrade.py +570 -0
  191. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/.github/agent.env +12 -0
  192. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/.github/workflows/promote-to-prod.yaml +371 -0
  193. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/.github/workflows/staging.yaml +450 -0
  194. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/argocd/application-dev.yaml +43 -0
  195. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/argocd/application-prod.yaml +41 -0
  196. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/argocd/application-staging.yaml +43 -0
  197. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/.helmignore +14 -0
  198. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/Chart.yaml +21 -0
  199. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/examples/networkpolicy.yaml +103 -0
  200. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/NOTES.txt +48 -0
  201. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/_helpers.tpl +189 -0
  202. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/certificate.yaml +15 -0
  203. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/configmap.yaml +10 -0
  204. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/deployment.yaml +199 -0
  205. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/hpa.yaml +22 -0
  206. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/httproute.yaml +30 -0
  207. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/ingress.yaml +39 -0
  208. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/networkpolicy.yaml +48 -0
  209. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/pdb.yaml +13 -0
  210. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/postgresql-secret.yaml +37 -0
  211. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/service.yaml +15 -0
  212. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/serviceaccount.yaml +13 -0
  213. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/servicemonitor.yaml +42 -0
  214. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values-dev.yaml +22 -0
  215. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values-prod.yaml +45 -0
  216. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values-staging.yaml +29 -0
  217. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values.yaml +396 -0
  218. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/tests/integration/test_chart.py +269 -0
  219. graph_agents_cli/scaffold/deployment_targets/none/README.md +5 -0
  220. graph_agents_cli/scaffold/deployment_targets/none/python/README.md +6 -0
  221. graph_agents_cli/scaffold/utils/__init__.py +13 -0
  222. graph_agents_cli/scaffold/utils/backup.py +212 -0
  223. graph_agents_cli/scaffold/utils/build_record.py +257 -0
  224. graph_agents_cli/scaffold/utils/cli_options.py +184 -0
  225. graph_agents_cli/scaffold/utils/fs.py +83 -0
  226. graph_agents_cli/scaffold/utils/generate_locks.py +214 -0
  227. graph_agents_cli/scaffold/utils/generation_metadata.py +88 -0
  228. graph_agents_cli/scaffold/utils/keyedit.py +768 -0
  229. graph_agents_cli/scaffold/utils/keymerge.py +537 -0
  230. graph_agents_cli/scaffold/utils/language.py +138 -0
  231. graph_agents_cli/scaffold/utils/lock_utils.py +94 -0
  232. graph_agents_cli/scaffold/utils/logging.py +77 -0
  233. graph_agents_cli/scaffold/utils/manifest.py +292 -0
  234. graph_agents_cli/scaffold/utils/merge.py +970 -0
  235. graph_agents_cli/scaffold/utils/merge3.py +216 -0
  236. graph_agents_cli/scaffold/utils/openapi_seed.py +199 -0
  237. graph_agents_cli/scaffold/utils/remote_template.py +376 -0
  238. graph_agents_cli/scaffold/utils/template.py +1352 -0
  239. graph_agents_cli/scaffold/utils/upgrade.py +894 -0
  240. graph_agents_cli/scaffold/utils/version.py +438 -0
  241. graph_agents_cli/secrets/__init__.py +15 -0
  242. graph_agents_cli/secrets/_apply.py +954 -0
  243. graph_agents_cli/secrets/_required.py +188 -0
  244. graph_agents_cli/secrets/cmd_secrets.py +211 -0
  245. graph_agents_cli/setup/__init__.py +13 -0
  246. graph_agents_cli/setup/_antigravity.py +221 -0
  247. graph_agents_cli/setup/cmd_auth.py +1030 -0
  248. graph_agents_cli/setup/cmd_dev_token.py +513 -0
  249. graph_agents_cli/setup/cmd_setup.py +428 -0
  250. graph_agents_cli/setup/cmd_update.py +140 -0
  251. graph_agents_cli/skills/__init__.py +13 -0
  252. graph_agents_cli/skills/_bundle.py +65 -0
  253. graph_agents_cli/skills/data/README.md +19 -0
  254. graph_agents_cli/skills/data/graph-agents-cli-deploy/SKILL.md +357 -0
  255. graph_agents_cli/skills/data/graph-agents-cli-deploy/references/github-settings.md +113 -0
  256. graph_agents_cli/skills/data/graph-agents-cli-deploy/references/gitops.md +137 -0
  257. graph_agents_cli/skills/data/graph-agents-cli-deploy/references/kubernetes.md +315 -0
  258. graph_agents_cli/skills/data/graph-agents-cli-deploy/references/secrets.md +160 -0
  259. graph_agents_cli/skills/data/graph-agents-cli-eval/SKILL.md +303 -0
  260. graph_agents_cli/skills/data/graph-agents-cli-eval/references/dataset_schema.md +282 -0
  261. graph_agents_cli/skills/data/graph-agents-cli-eval/references/metrics-guide.md +143 -0
  262. graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/SKILL.md +659 -0
  263. graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/references/langchain-models.md +124 -0
  264. graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/references/langgraph.md +235 -0
  265. graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/references/template-contract.md +477 -0
  266. graph_agents_cli/skills/data/graph-agents-cli-observability/SKILL.md +231 -0
  267. graph_agents_cli/skills/data/graph-agents-cli-observability/references/langsmith.md +46 -0
  268. graph_agents_cli/skills/data/graph-agents-cli-observability/references/otel.md +59 -0
  269. graph_agents_cli/skills/data/graph-agents-cli-scaffold/SKILL.md +414 -0
  270. graph_agents_cli/skills/data/graph-agents-cli-scaffold/references/flags.md +134 -0
  271. graph_agents_cli/skills/data/graph-agents-cli-workflow/SKILL.md +478 -0
  272. graph_agents_cli/skills/data/graph-agents-cli-workflow/references/brainstorming.md +118 -0
  273. graph_agents_cli/skills/data/graph-agents-cli-workflow/references/commands.md +419 -0
  274. graph_agents_cli/skills/data/graph-agents-cli-workflow/references/extension.md +156 -0
  275. graph_agents_cli/skills/data/graph-agents-cli-workflow/references/internals.md +67 -0
  276. graph_agents_cli/skills/data/graph-agents-cli-workflow/references/spec-template.md +56 -0
  277. graph_agents_cli/skills/data/graph-agents-cli-workflow/references/terminology.md +119 -0
  278. graph_agents_cli/system/__init__.py +15 -0
  279. graph_agents_cli/system/_apply.py +519 -0
  280. graph_agents_cli/system/_checks.py +1023 -0
  281. graph_agents_cli/system/_deploy.py +215 -0
  282. graph_agents_cli/system/_model.py +363 -0
  283. graph_agents_cli/system/_system.py +664 -0
  284. graph_agents_cli/system/_views.py +208 -0
  285. graph_agents_cli/system/cmd_system.py +423 -0
  286. graph_agents_cli-0.3.1.dist-info/METADATA +162 -0
  287. graph_agents_cli-0.3.1.dist-info/RECORD +291 -0
  288. graph_agents_cli-0.3.1.dist-info/WHEEL +4 -0
  289. graph_agents_cli-0.3.1.dist-info/entry_points.txt +2 -0
  290. graph_agents_cli-0.3.1.dist-info/licenses/LICENSE +201 -0
  291. graph_agents_cli-0.3.1.dist-info/licenses/NOTICE +19 -0
@@ -0,0 +1,571 @@
1
+ # {{cookiecutter.project_name}}
2
+
3
+ A LangGraph agent scaffolded by graph-agents-cli.
4
+
5
+ | Setting | Value |
6
+ |---|---|
7
+ | Runtime | `{{cookiecutter.runtime}}` |
8
+ | Model | `{{cookiecutter.model_provider}}` / `{{cookiecutter.model}}` (env-driven, see `.env.example`) |
9
+ | Deployment target | `{{cookiecutter.deployment_target}}` |
10
+ | CD mode | `{{cookiecutter.cd}}` |
11
+ | Auth policy | `{{cookiecutter.auth_policy}}` |
12
+ | Checkpointer (deployed) | `{{cookiecutter.checkpointer}}` |
13
+
14
+ ## Quick start
15
+
16
+ ```bash
17
+ cp .env.example .env # set {{cookiecutter.provider_key_var}} (or MODEL_PROVIDER=fake to try it without a key)
18
+ graph-agents-cli login --write-env # checks the setup; prompts for missing keys{% if cookiecutter.auth_policy == 'shared-bearer' %}, generates API_KEY{% endif %}
19
+ graph-agents-cli install # uv sync from the committed uv.lock
20
+ {%- if cookiecutter.auth_policy == 'jwt' %}
21
+ export GRAPH_AGENTS_CLI_API_KEY="$(graph-agents-cli auth dev-token --sub alice --roles user)"
22
+ {%- endif %}
23
+ graph-agents-cli run "What's the weather in San Francisco?" # the example tool (tools/weather.py)
24
+ graph-agents-cli eval run # generate traces, grade them, enforce the gate
25
+ graph-agents-cli playground # http://127.0.0.1:8000/playground (APP_ENV=dev)
26
+ ```
27
+
28
+ {%- if cookiecutter.auth_policy == 'shared-bearer' %}
29
+ `API_KEY` is the shared bearer key every client sends (`Authorization: Bearer ...`); the local
30
+ server answers 503 until it is set. `login --write-env` generates one, or run
31
+ `python -c "import secrets; print(secrets.token_hex(32))"`. `run` and `eval` send the `API_KEY`
32
+ in `.env`; for a deployed agent put its key in `GRAPH_AGENTS_CLI_API_KEY` (kept out of argv and
33
+ shell history, unlike `--header`).
34
+ {%- elif cookiecutter.auth_policy == 'jwt' %}
35
+ Every request needs a JWT the server can verify. For local runs without an identity provider,
36
+ `graph-agents-cli auth dev-token --sub <user> [--roles r1,r2]` creates a dev key pair in
37
+ `.graph-agents-cli/dev-jwt/` (git ignored), fills the blank `AUTH_JWT_PUBLIC_KEY`,
38
+ `AUTH_JWT_ISSUER` and `AUTH_JWT_AUDIENCE` in `.env`, and prints a token; it refuses unless
39
+ `APP_ENV=dev`. `run` and `eval` send whatever `GRAPH_AGENTS_CLI_API_KEY` holds as the bearer
40
+ token, which keeps it out of argv and shell history (do not pass tokens with `--header`). Mint one
41
+ token per test user to exercise thread ownership and roles; restart a kept server
42
+ (`graph-agents-cli run --stop-server`) after the first `dev-token`. Deployed environments verify
43
+ tokens from your identity provider (`AUTH_JWT_JWKS_URL`, see Authentication below); never deploy
44
+ the dev key.
45
+ {%- else %}
46
+ Authentication follows `AUTH_POLICY={{cookiecutter.auth_policy}}` (see Authentication below): until
47
+ `{{cookiecutter.agent_directory}}/policies/custom.py` is implemented every request gets 503. Send
48
+ what your policy reads with `run --header 'Name: value'` or `--cookie name=value`.
49
+ {%- endif %}
50
+ Local development needs no database: `.env.example` sets `CHECKPOINTER=memory`. The example tool
51
+ and the eval dataset are starting points: replace or delete `{{cookiecutter.agent_directory}}/tools/weather.py`
52
+ and its eval case when you write your own; the tests under `tests/` do not depend on either.
53
+
54
+ ## Layout
55
+
56
+ ```
57
+ {{cookiecutter.agent_directory}}/
58
+ ├── agent.py # exports `graph` (compiled LangGraph agent, no checkpointer bound)
59
+ ├── fast_api_app.py # exports `app`: the HTTP API below
60
+ ├── app_utils/ # auth, api_client, approvals, chat, threads, db, limits, metrics, middleware, model, telemetry, a2a
61
+ ├── policies/ # AuthPolicy implementations (custom.py is a fail-closed stub)
62
+ └── tools/ # every module declares API_CALLS and TOOLS
63
+ tests/{unit,integration,eval,load_test}
64
+ {%- if cookiecutter.deployment_target == 'kubernetes' %}
65
+ deployment/helm/{{cookiecutter.project_name}}/ # chart, values.yaml, values-{dev,staging,prod}.yaml
66
+ {%- if cookiecutter.cd == 'argocd' %}
67
+ deployment/argocd/ # application-{dev,staging,prod}.yaml
68
+ {%- endif %}
69
+ {%- endif %}
70
+ .github/ # workflows, agent.env (their settings), CODEOWNERS
71
+ langgraph.json # graph, custom app and auth handler (LangGraph Studio / Server)
72
+ api-policy.yaml # when present: the external APIs tools may call, and how (`api show`)
73
+ Dockerfile # {{cookiecutter.runtime}} image (runs as uid 1000)
74
+ .env.example # the full environment contract, with defaults
75
+ graph-agents-cli-manifest.yaml
76
+ ```
77
+
78
+ ## Commands
79
+
80
+ | Command | Purpose |
81
+ |---|---|
82
+ | `graph-agents-cli playground` | Run the app with reload; `--graph` opens LangGraph Studio (bypasses the auth policy) |
83
+ | `graph-agents-cli run "prompt" [--mode a2a] [--url URL] [--thread-id ID]` | One-shot chat; `--url` targets a deployed agent. A bearer credential goes in `GRAPH_AGENTS_CLI_API_KEY` (locally and with `--url`), never on the command line |
84
+ {%- if cookiecutter.auth_policy == 'jwt' %}
85
+ | `graph-agents-cli auth dev-token --sub USER [--roles R,...] [--ttl 12h]` | A token for local runs (dev key in `.graph-agents-cli/dev-jwt/`, `APP_ENV=dev` only) |
86
+ {%- endif %}
87
+ | `uv run pytest` | Unit and integration tests with the deterministic `fake` model and the in-memory checkpointer (`TEST_POSTGRES_DSN` opts the Postgres tests in) |
88
+ | `graph-agents-cli eval run` | Generate traces (`artifacts/traces/`) and grade them against the gate in `tests/eval/eval_config.yaml` |
89
+ | `graph-agents-cli lint` | ruff plus the API-policy check of every tool's `API_CALLS` |
90
+ | `graph-agents-cli api show` / `api check` | The effective outbound API policy and every tool's declared calls / the policy check alone |
91
+ | `graph-agents-cli api add\|access\|allow\|deny\|revoke\|limits\|remove ...` | Change `api-policy.yaml` (diff first, `--dry-run` to preview; see Outbound API access) |
92
+ | `graph-agents-cli build` | `docker build` with the runtime's Dockerfile |
93
+ {%- if cookiecutter.deployment_target == 'kubernetes' %}
94
+ | `graph-agents-cli infra check --env <env>` | Read-only report of cluster, GitHub and placeholder prerequisites |
95
+ | `graph-agents-cli secrets apply --env <env>` | Create or update the environment's app Secret from the allow-listed keys of `.env.<env>` |
96
+ | `graph-agents-cli secrets status --env <env>` | Which keys the Secret holds (never values); exit 1 when a required key is missing |
97
+ | `graph-agents-cli deploy --env <env> [--image REF] [--status] [--restart] [--dry-run]` | Deploy per the CD mode (see below) |
98
+ {%- endif %}
99
+
100
+ ## The API
101
+
102
+ | Route | Behaviour |
103
+ |---|---|
104
+ | `POST /chat` | `{"thread_id": "optional", "message": "...", "metadata": {}}` with `Accept: text/event-stream`; streams `message.start`, `message.delta`, `tool.call`, `tool.result`, `message.end` (usage, latency, status) or `error`. Omit `thread_id` to start a thread (the server generates a random id); send it to continue one. A run that pauses for approval ends with `message.end` status `awaiting_approval` and `approval`; while it waits, a new message gets 409 `{"code": "approval_pending"}`. With a response schema (Structured answers below), `message.end` carries the answer as `structured_response` |
105
+ | `GET /threads` | The caller's threads, most recent first (`?limit=1..100&offset=`), each with its `owner` hashed; `?scope=all` lists every principal's, for a role in `AUTH_READ_ACROSS_ROLES` only |
106
+ | `GET /threads/{id}/messages` | A thread's messages (owner, or a role in `AUTH_READ_ACROSS_ROLES`) |
107
+ | `GET /threads/{id}/approvals` | The thread's approvals of gated API calls (owner and read-across roles: all; an approver: the ones it may decide) |
108
+ | `POST /threads/{id}/approvals/{approval_id}` | `{"decision": "approve" \| "reject", "comment": "..."}` by an approver; streams the resumed run with the `/chat` events. 403 not an approver, 409 decided already, 410 expired |
109
+ | `GET /approvals` | Approvals across threads: the caller's own and the ones a role of theirs may decide (`?status=pending\|approved\|rejected\|expired&limit=&offset=`) |
110
+ | `DELETE /threads/{id}` | Delete a thread, its checkpoints, run records, approvals and A2A tasks (owner only; 409 while a run is in progress) |
111
+ | `GET /health` | Liveness: `{"status": "ok", "runtime", "checkpointer"}` (no auth) |
112
+ | `GET /ready` | Readiness: 200 when the database is set up and answers within 2 s, else 503 (no auth) |
113
+ | `GET /metrics` | Prometheus text (no auth unless `METRICS_TOKEN` is set; `METRICS_ENABLED=false` turns it off) |
114
+ | `/a2a/{{cookiecutter.agent_directory}}` | A2A JSON-RPC; card at `/a2a/{{cookiecutter.agent_directory}}/.well-known/agent-card.json` (description `A2A_DESCRIPTION`, version `AGENT_VERSION`); tasks are private to their principal and kept for `A2A_TASK_TTL_S` after their last update, in the app's database when it has one (table `a2a_tasks`{% if cookiecutter.runtime == 'langgraph-server' %}; `agent_a2a_tasks` in the server's `DATABASE_URI`{% endif %}: every replica sees them, and they survive restarts), else in process memory (an unknown task is -32001, A2A 0.3 included); `SendMessage` returns the reply as one text part |
115
+ | `/playground`, `/docs`, `/openapi.json` | Only under `APP_ENV=dev` |
116
+ {%- if cookiecutter.runtime == 'langgraph-server' %}
117
+
118
+ Under LangGraph Server these routes are mounted beside the native Assistants/Threads/Runs API
119
+ (`langgraph.json` `http.app`) and the same policy is the server's auth handler (`langgraph.json`
120
+ `auth`). Thread ids must be UUIDs, and `DELETE /threads/{id}` is the server's own route with the
121
+ same owner rule. Assistants, crons and store writes need a role in `AUTH_ADMIN_ROLES`; every
122
+ other native action a handler does not allow is denied. The server image disables the server's
123
+ unauthenticated `/docs`, `/openapi.json`, `/info` and `/metrics`.
124
+ {%- endif %}
125
+
126
+ Behaviour and its settings (defaults in `.env.example`; a value that does not parse stops the app
127
+ at startup):
128
+
129
+ - **One run per thread:** a second `/chat` on a busy thread gets 409 `{"code": "thread_busy"}`.
130
+ Under postgres the lock is a lease every replica honours: a replica that dies frees its
131
+ threads 30 s later, and a run that can no longer renew its lease stops before it writes.
132
+ - **Guardrails:** a run is cancelled after `RUN_TIMEOUT_S` (300); each model request has
133
+ `MODEL_TIMEOUT_S` (60) and `MODEL_MAX_RETRIES` (2). OpenAI-API models also take
134
+ `MODEL_REASONING_EFFORT` and `MODEL_USE_RESPONSES_API` (`true`: the Responses API, which
135
+ some models need for tools; unset: langchain-openai chooses). `RECURSION_LIMIT` (50, room for 24
136
+ sequential tool calls) caps graph steps: a run that reaches it ends with a reply saying so
137
+ (`message.end` status `step_limit`) and keeps its work in the thread. A client disconnect
138
+ cancels the run. Idle streams get a keep-alive comment every `SSE_HEARTBEAT_S` (15).
139
+ - **Valid history:** a run stopped mid tool call (a timeout, a crash, an outage) leaves a call
140
+ without a result; the next run answers it with an error result right after the call before
141
+ adding its turn, so model providers accept the thread. A tool call whose arguments are not
142
+ valid JSON (some OpenAI-compatible models return them) runs no tool: the agent answers it
143
+ with an error result and asks the model again, at most twice (`AnswerInvalidToolCalls`).
144
+ - **Run records:** written as `running` when a run starts and updated when it ends (`ok`,
145
+ `awaiting_approval`, `step_limit`, `error`, `timeout`, `cancelled`, `interrupted`); runs of a
146
+ process that died are marked `interrupted` within about a minute.
147
+ - **Limits:** bodies over `MAX_REQUEST_BYTES` get 413; a message over `MAX_MESSAGE_CHARS`
148
+ (32 000) gets 422 on `/chat` and an invalid-params error over A2A; metadata beyond
149
+ `MAX_METADATA_KEYS` / `MAX_METADATA_VALUE_CHARS` gets 422. A 422 never echoes the submitted
150
+ values.
151
+ - **Thread ids** are shared by every caller: an id another principal used first is theirs
152
+ (403). Let the server generate ids, or use unguessable ones (UUID4).
153
+ - **Errors:** the `error` event is `{"code", "message", "error_id", "run_id"}`; an unhandled error
154
+ answers 500 with an `error_id`. Details are only in the server log under that id. A failed
155
+ tool call's `tool.result` carries an `error_id` and, outside `APP_ENV=dev`, a generic text
156
+ (the error text is for the model only). Every response carries `X-Request-ID`. An
157
+ unreachable database answers 503 within a few seconds and logs one line.
158
+ - **Retention:** `RETENTION_DAYS=N` deletes threads idle for more than N days (hourly; 0 keeps
159
+ everything).
160
+ {%- if cookiecutter.runtime == 'langgraph-server' %}
161
+ - **Logging:** LangGraph Server formats the lines (its `LOG_JSON` and `LOG_LEVEL`). Its access
162
+ lines lose their `query_string` field; outbound API calls are logged by API, method,
163
+ operation id and path template, never their values (`httpx` and `httpcore` log at WARNING
164
+ only); warnings are records too, with the values pydantic echoes redacted.
165
+ {%- else %}
166
+ - **Logging:** JSON lines outside `APP_ENV=dev` (`LOG_FORMAT`, `LOG_LEVEL`) with request id, run
167
+ id, thread id and a hashed principal (HMAC-keyed with `PRINCIPAL_HASH_SALT` when set). Access
168
+ lines drop query strings; outbound API calls are logged by API, method, operation id and path
169
+ template, never their values (`httpx` and `httpcore` log at WARNING only); warnings are JSON
170
+ records too, with the values pydantic echoes redacted.
171
+ {%- endif %}
172
+ - **CORS:** off unless `CORS_ALLOW_ORIGINS` lists origins.
173
+ - **Database:** one health-checked pool per process (`DB_POOL_MIN_SIZE`, `DB_POOL_MAX_SIZE`);
174
+ connections get `connect_timeout=5` and TCP keepalives unless the DSN sets them. The app
175
+ starts even while Postgres is unreachable (`/ready` 503 until it answers) and is ready again
176
+ seconds after Postgres is.
177
+
178
+ No inbound rate limiting is built in: configure it at the gateway or ingress (outbound calls
179
+ can be limited per API, see below).
180
+
181
+ **Tool results are untrusted input.** Text a tool returns (a customer's note, an upstream error)
182
+ can carry instructions meant to steer the agent into acting on someone else's records.
183
+ `agent.py` fences every tool result the model reads (`UntrustedToolResults`) and its prompt
184
+ forbids following instructions found there; write tools should also call
185
+ `require_user_mentioned(record_id, runtime)` and, under a per-user auth policy,
186
+ `require_owner(owner_id, context=runtime.context)` (from `app_utils.api_client`), and
187
+ write-capable APIs should authorize the user themselves (`auth: forward`). This lowers the risk
188
+ without removing it. The control that holds is a person's approval of each write before it is
189
+ sent: gate write methods or operations with `approval` in `api-policy.yaml` (see Human approval of
190
+ calls below).
191
+
192
+ ## Structured answers
193
+
194
+ Put a JSON Schema whose root is an object in `{{cookiecutter.agent_directory}}/response_schema.json` and the agent answers in
195
+ that shape: `agent.py` builds it with `response_format=response_format(model, tools)` (`app_utils/structured.py`).
196
+ `RESPONSE_FORMAT_STRATEGY` picks how the model is made to: `auto` (default: the provider's own structured output
197
+ where the model has it and its client can send the schema, strict on OpenAI, else a `final_answer` tool the model
198
+ must call), `provider` or `tool`. Anthropic's client refuses a type list (`["string", "null"]`) and an `enum` with
199
+ no `type`: give every schema a `type` and write a nullable value as `anyOf` with `{"type": "null"}`.
200
+ Every answer is checked against the schema; one that does not fit goes back to the model (3 tries), then the run
201
+ ends with the `error` code `invalid_structured_response`. A completed run's only `message.delta` is the answer's
202
+ JSON text and `message.end` carries the object as `structured_response`; the A2A reply adds a data part with it;
203
+ `expect.json_schema` in an eval case checks it as it is. The schema may use `type`, `enum`, `const`, `properties`,
204
+ `required`, `additionalProperties`, the size and number bounds, `pattern`, `anyOf`/`oneOf`/`allOf`/`not` and
205
+ local `$ref`s; anything else stops startup (and `graph-agents-cli lint`). On OpenAI Chat Completions strict mode
206
+ makes every property required (give one that may be empty a `null` type) and every tool strict.
207
+ The tests run with `RESPONSE_SCHEMA_PATH=none` (text answers, `tests/conftest.py`), whatever shape the project
208
+ declares; `tests/unit/test_structured.py` checks the project's schema and that `agent.py` answers in it (skipped
209
+ when the fake model's text cannot match the schema's `pattern`).
210
+
211
+ ## Model and judge
212
+
213
+ `MODEL_PROVIDER` / `MODEL_NAME` (and `OPENAI_BASE_URL` for `openai-compatible`) select the agent model
214
+ through LangChain's `init_chat_model`; the provider key lives in `{{cookiecutter.provider_key_var}}`. The eval judge
215
+ uses `JUDGE_MODEL_PROVIDER`, `JUDGE_MODEL_NAME`, `JUDGE_BASE_URL`, `JUDGE_API_KEY` and defaults to the agent's
216
+ values. Selecting a hosted provider sends prompts, tool results and context to that provider: decide what may
217
+ leave and publish a privacy notice before connecting one.
218
+
219
+ ## Outbound API access
220
+
221
+ Tools reach external APIs only through `get_client("<api>")` of
222
+ `{{cookiecutter.agent_directory}}/app_utils/api_client.py`, which enforces `api-policy.yaml` and refuses, before
223
+ sending, any API, method or operation outside it. It fails closed: without the file every outbound call is
224
+ refused. The client sends every method the policy allows (`request()`, or `get`, `post`, `put`, `patch`,
225
+ `delete`, `head`, `options`) with a JSON body, query parameters and headers.
226
+
227
+ Each API declares `base_url_env` (the URL may carry a path prefix), `auth` (`none`, `bearer` with
228
+ `token_env`, `forward`, which sends the caller's own `attributes["credentials"][<api>]` or, with
229
+ `forward_audience`, the caller's own token when it was minted for that audience too, or `exchange`, which
230
+ sends a token the issuer mints for `exchange.audience` in exchange for the caller's own, RFC 8693, set up
231
+ with the `TOKEN_EXCHANGE_*` variables, and never sends one that names no actor unless
232
+ `exchange.allow_actorless: true`; neither is available under langgraph-server, which would persist
233
+ the caller's credentials), the required `allowed_methods`, and optional
234
+ `allowed_operations` / `denied_operations` (an allowed entry pinning both `operationId` and `path` needs
235
+ both to match), `openapi`, `timeouts_ms`, `pagination` (`max_page_size` is enforced for every spelling of the
236
+ parameter) and `limits`: `max_calls_per_run` (calls to that API within one agent run) and
237
+ `rate_per_minute` (a token bucket per process, so per replica). A call over a limit is refused before it is
238
+ sent, with a reason the model can read. `approval` names the calls a person must approve before they are
239
+ sent (see Human approval of calls). Unknown and repeated keys are errors, so a typo never widens access. Denials
240
+ win and hold on the endpoint: a denial pinning a path refuses every call to it whatever `operation_id` the
241
+ call gives, and a call that leaves out what a denial knows the operation by is refused by it, so a denial by
242
+ `operationId` alone refuses every call without `operation_id` (name it on the call and in `API_CALLS`), but
243
+ only knows that label: pin the denial's `path` too. With `openapi`, `lint` also refuses a declared
244
+ `operation_id` the spec does not give that method and path. Paths match after decoding percent-encoded
245
+ unreserved characters and ignoring one trailing slash; letter case counts for allows and is ignored for
246
+ denials and approval gates, which also cover a literal segment's dot-suffixed spellings (a denial of
247
+ `/orders/{order_id}/cancel` refuses `/orders/7/cancel.json` and `cancel.`, which servers that route format
248
+ suffixes or drop a trailing dot send to the same endpoint). A `;` in a request path is refused, and so is a
249
+ segment with a control character or with whitespace at either end or next to a dot, also percent-encoded
250
+ (`cancel%20`, `cancel%20.json`, `7%00`, which servers that trim segments or the name before a suffix, or end
251
+ a path at a NUL, route elsewhere); `lint` refuses the same in declared paths, and an encoded slash,
252
+ backslash, `;` or dot segment too. Pass model input as `path_params` of a declared template, never as part
253
+ of a concrete path.
254
+ `graph-agents-cli api show` lists the APIs `api-policy.yaml` declares now, what each allows, and every
255
+ tool's declared calls; without the file, declare the first API with `graph-agents-cli api add` (below). When
256
+ `{{cookiecutter.agent_directory}}/tools/example_api.py` exists (a project created with a policy), it shows
257
+ the pattern with one call the policy allows: replace it with your own.
258
+ Every `*.py` under `{{cookiecutter.agent_directory}}/tools/` (subpackages included) declares `API_CALLS` as one
259
+ module-level literal list; `graph-agents-cli lint` fails on an undeclared or disallowed call (and prints the
260
+ `graph-agents-cli api` command that would allow it), and on `API_CALLS` changed anywhere else (`+=`,
261
+ `.append()`, a conditional assignment), because it cannot read those calls.
262
+
263
+ ### Human approval of calls
264
+
265
+ An API's `approval` block holds the calls it names until a person approves them. It is the control for
266
+ write actions and for instructions planted in data the agent reads: whatever the model was talked into, the
267
+ request waits for someone who sees exactly what it does.
268
+
269
+ ```yaml
270
+ approval:
271
+ required_for: # at least one of:
272
+ methods: [POST, PATCH, PUT, DELETE] # these methods ("*": every one), and/or
273
+ operations: # entries shaped like allowed_operations
274
+ - operationId: cancelOrder
275
+ path: /orders/{order_id}/cancel
276
+ approvers: [requester] # "requester" and/or "role:<name>"
277
+ timeout_s: 900 # 30..86400 (default 900); unanswered in time = rejected
278
+ ```
279
+
280
+ - **What it gates.** A call is gated when its method is in `required_for.methods` or an entry of
281
+ `required_for.operations` covers it (a `path` gates every call to that path whatever `operation_id` it
282
+ names). Approval never widens access: a gated call must still pass `allowed_methods`, the allowed and
283
+ denied operations and the limits, and a denial still wins. Gate the writes that matter (every write
284
+ method, or the operations that act on other people's records); reads usually need no gate.
285
+ - **Other approvers for other calls.** `approval` may be a list of rules of that shape, each with its own
286
+ `required_for`, `approvers` and `timeout_s` (the requester confirms changes to their orders; a
287
+ `role:admin` approves new orders). The **first** rule in file order whose `required_for` covers a call
288
+ gates it, with that rule's approvers, which are recorded with the approval and decide it; a later rule
289
+ that also covers the call does not apply to it. Put narrow rules first, with `path` and `methods` pinned in
290
+ their entries, and name `operation_id` on every call: an entry by `operationId` alone cannot rule out a call
291
+ that names none, so such a call that a later rule with other approvers also covers is refused (it could be
292
+ either rule's). `graph-agents-cli api approval NAME --add-rule ...` adds a rule and `--rule N` changes one
293
+ (`approval[N]`, from 0); `lint` and `graph-agents-cli api show` name the rule each declared call waits for.
294
+ - **Who approves.** `requester` lets the principal who started the run (the thread's owner) confirm it;
295
+ `role:<name>` lets any principal holding that role decide, never the requester itself (four eyes), unless
296
+ `requester` is listed too. For example `approvers: [requester]` asks the user to confirm each order
297
+ change; `approvers: ["role:support-lead"]` makes a second person approve every refund. Deciding also needs
298
+ the auth policy's `approval.decide` action. With `shared-bearer` every caller is the one principal
299
+ `shared` (the requester of every run), so `role:` approvers need a per-user policy (`jwt` or `custom`).
300
+ - **How it runs.** The run pauses before anything is sent and ends its stream with `message.end` status
301
+ `awaiting_approval` and `approval`: `approval_id`, `api`, `method`, `path`, `query`, `body`,
302
+ `operation_id`, `tool`, `reason` (the tool and the text the model wrote with the call), `approvers`,
303
+ `expires_at` (`approvals` lists every one when parallel calls wait). The thread takes no new message
304
+ meanwhile (409 `approval_pending`). An approver lists it (`GET /threads/{id}/approvals`, or
305
+ `GET /approvals?status=pending` across threads) and decides it
306
+ (`POST /threads/{id}/approvals/{approval_id}`); the run resumes, acting as the requester, and streams
307
+ as `/chat` does, to the decider (a `role:` approver sees that tool result and reply). A2A clients see the task move to `input-required` with the approval in a data part and
308
+ answer on the same task with a data part `{"approval_id": "...", "decision": "approve"}` (the same checks,
309
+ `approval.decide` included; a task belongs to its principal, so only the requester decides there). The
310
+ playground (`APP_ENV=dev`) shows Approve and Reject buttons. `graph-agents-cli run` and
311
+ `graph-agents-cli approvals list|approve|reject` do the same from the command line.
312
+ - **Bound and single use.** The approval covers the exact request: API, method, URL with the rendered path,
313
+ query, JSON body, operation id and the tool's own headers (a SHA-256 of them). On resume the tool runs
314
+ again and the client sends the request only when it is the same one, then marks the approval used, so it is
315
+ sent once and never replayed. A decision is bound to the call it was taken for (its API, method and
316
+ path), not to the policy of the moment: a rejected or expired call is never sent, even if a new policy no
317
+ longer gates it; an approved one is sent only while the policy still allows it (a later denial or a
318
+ narrower `allowed_methods`/`allowed_operations` refuses it) and still gates it with the same approvers
319
+ (with rules: the rule that gates it now);
320
+ and a call still waiting when another call's decision resumes the run waits on for its own approval (if
321
+ the policy now refuses it, it is refused and its approval expires). The approvals table binds the decision
322
+ to its tool call too: a tool call that runs again without a decision (a run continued without input or
323
+ replayed from a checkpoint through LangGraph Server's own API, or a copied thread) does not send a call an
324
+ approval was asked for; a rejected, expired or pending one is refused and an approved one is sent only by
325
+ the run its decision resumed, once. Keep the agent's middleware (`agent.middleware()`), which names the
326
+ tool call. Anything else sends nothing and the tool gets an error saying why, which the model relays. A
327
+ pending approval that expires is closed on the next decision or message on the thread; one a failed or
328
+ cancelled run leaves behind, or a resumed run no longer waits for, is expired when that run ends.
329
+ - **Writing a tool that makes a gated call.** Make at most one gated call per tool call: on resume the tool
330
+ runs again from its start, so a second gated call in the same tool call is refused after the first was
331
+ sent, and anything the tool does before the gated call runs again (keep other side effects after it).
332
+ `client.request(..., redact=["card_number"])` masks fields the approver need not read (the request is
333
+ bound as sent). An `auth: forward` call approved by someone other than the requester is not sent: the
334
+ requester's credential is never stored.
335
+ - **Records.** Approvals live in the app's database (table `approvals`{% if cookiecutter.runtime == 'langgraph-server' %}; `agent_approvals` in the server's `DATABASE_URI` database{% endif %}): the call,
336
+ the requester and decider hashed, the decision, its comment and time, and when it was used. Once decided
337
+ or expired, the query and body are dropped unless `TRACE_CAPTURE=full`. Deleting a thread deletes its
338
+ approvals. `/metrics` counts `agent_approvals_total{event="requested|approved|rejected|expired"}`.
339
+ {%- if cookiecutter.runtime == 'langgraph-server' %}
340
+ The local `langgraph dev` server keeps its threads in `.langgraph_api/` across a restart or a hot reload
341
+ (a code change), and the approvals with them, in `.langgraph_api/agent_approvals.json` (written before
342
+ each change takes effect, so every binding above holds after a reload); a file that cannot be read stops
343
+ its startup. Delete `.langgraph_api/` to reset both; it stays out of git and images.
344
+ {%- endif %}
345
+ {%- if cookiecutter.runtime == 'langgraph-server' %}
346
+ - **LangGraph Server.** The run pauses and resumes through the server's own interrupt and resume. The
347
+ server's auth handler refuses a run that carries a `command` (a resume) from outside the app, and the
348
+ client sends only an approval the app recorded as approved, so the native API cannot skip the decision.
349
+ A new native run on a thread whose approval is pending gets 409, as `/chat` does, and a run without input
350
+ or from a checkpoint (which would run a paused step's tool calls again) gets 403 on a thread that has
351
+ approvals or waits on a gated call; a thread that has approvals is not copied (403).
352
+ Resuming needs the in-process loopback (`LANGGRAPH_SERVER_URL` unset, the default).
353
+ {%- endif %}
354
+
355
+ ### Changing the policy
356
+
357
+ `api-policy.yaml` belongs to this project and evolves with the agent; `scaffold upgrade` and `enhance` never
358
+ touch it. There is no default access level: every API lists its methods explicitly.
359
+
360
+ | Command | Change |
361
+ |---|---|
362
+ | `graph-agents-cli api add NAME --base-url-env ENV --auth none\|bearer\|forward\|exchange [--token-env ENV] [--audience AUD] --access read-only\|read-write\|custom [--methods M,...] [--openapi SPEC] [--max-calls-per-run N] [--rate-per-minute N]` | Declare an API; `--access` is required. read-only = GET, HEAD; read-write = GET, HEAD, POST, PUT, PATCH, DELETE; custom = `--methods` |
363
+ | `graph-agents-cli api access NAME read-only\|read-write\|custom [--methods M,...]` | Set the allowed methods |
364
+ | `graph-agents-cli api allow NAME OPERATION_ID [--method M --path P]` (or `--method M --path P`) | Add an `allowed_operations` entry pinning every field given (with `openapi`, the id must exist and its method and path are filled in). Creating the list narrows access to the listed operations: the command says so |
365
+ | `graph-agents-cli api deny NAME OPERATION_ID [--method M --path P]` (or `--method M --path P`) | Add a `denied_operations` entry (pin the path: it then holds whatever `operation_id` a call gives) |
366
+ | `graph-agents-cli api revoke NAME OPERATION_ID [--from allowed\|denied]` | Remove matching entries |
367
+ | `graph-agents-cli api limits NAME [--max-calls-per-run N\|none] [--rate-per-minute N\|none]` | Set or clear limits |
368
+ | `graph-agents-cli api approval NAME [--methods M,...] [--operations OP,...] [--approvers requester,role:R] [--timeout-s N] [--remove] [--dry-run]` | Set or remove the API's `approval` gate (`--approvers` is required for a new one) |
369
+ | `graph-agents-cli api remove NAME` | Remove an API |
370
+
371
+ Each command validates the result with the rules the agent enforces, prints a diff (comments and key order
372
+ are kept), keeps the manifest (`secrets.keys`), `.env.example` and the chart's `values.yaml` in step, and
373
+ writes atomically; `--dry-run` shows the diff only. Widening access (more methods or operations, a lifted
374
+ denial, a raised limit, a removed or loosened `approval` gate) is a reviewed change: `.github/CODEOWNERS`
375
+ covers `api-policy.yaml`. Narrowing is
376
+ always safe, and the runtime keeps refusing anything outside the policy even if a tool declares otherwise.
377
+
378
+ Adding functionality to a working agent, for example letting it update orders:
379
+
380
+ 1. Change the policy, reviewing each printed diff (`--dry-run` first). If `orders` has no
381
+ `allowed_operations` yet, every operation within its methods is allowed: first
382
+ `graph-agents-cli api allow orders <operation> --method M --path P` for each operation the agent already
383
+ calls, because the first `allow` creates the list and every call not on it is refused from then on (the
384
+ command names the declared calls that become refused). Then
385
+ `graph-agents-cli api allow orders updateOrder --method PATCH --path /orders/{order_id}`,
386
+ and `graph-agents-cli api access orders custom --methods <the current methods>,PATCH` if PATCH is not
387
+ allowed yet. With the list in place the new method reaches only the listed operations; `api access`
388
+ without a list would allow every PATCH operation of the API.
389
+ 2. Write the tool with `{"api": "orders", "method": "PATCH", "operation_id": "updateOrder", "path": ...}` in
390
+ `API_CALLS`, calling `get_client("orders")`.
391
+ 3. `graph-agents-cli api check` (or `lint`), then add eval cases in `tests/eval/datasets/` and run
392
+ `graph-agents-cli eval run`.
393
+ 4. Open a pull request: CODEOWNERS approves the policy change.
394
+ 5. Build and deploy: the policy is baked into the image, so what passed staging is exactly what reaches
395
+ production; only base URLs (the chart's `env`) and tokens (the Secret) differ per environment.
396
+
397
+ ## Authentication
398
+
399
+ One policy (`AUTH_POLICY`) guards `/chat`, the thread routes and A2A{% if cookiecutter.runtime == 'langgraph-server' %}, and the server's native API{% endif %}.
400
+ An unknown policy never starts, and a misconfigured one stops the app outside `APP_ENV=dev` (under dev
401
+ requests get 503 and the problem is logged).
402
+
403
+ - `shared-bearer` (default): `Authorization: Bearer <API_KEY>`, compared in constant time; every caller is the
404
+ same principal, so use it for trusted callers only.
405
+ - `jwt`: each user gets their own principal from a verified OIDC/JWT bearer token. Set
406
+ `AUTH_JWT_JWKS_URL` (https outside dev) or `AUTH_JWT_PUBLIC_KEY`, `AUTH_JWT_ISSUER` and
407
+ `AUTH_JWT_AUDIENCE` (both required outside dev); optionally `AUTH_JWT_ALGORITHMS` (default `RS256,ES256`;
408
+ HS* only with `AUTH_JWT_ALLOW_HS=true` and a 32-byte `AUTH_JWT_SECRET` in the Secret),
409
+ `AUTH_JWT_PRINCIPAL_CLAIM` (`sub`), `AUTH_JWT_ROLES_CLAIM` (`roles`; dotted paths such as
410
+ `realm_access.roles`), `AUTH_JWT_LEEWAY_S` (60), `AUTH_JWT_JWKS_CACHE_S` (300). A missing or invalid token
411
+ gets 401 with a `WWW-Authenticate` challenge; unreachable issuer keys (after a 1 hour grace) get 503.
412
+ - `custom`: a fail-closed stub in `{{cookiecutter.agent_directory}}/policies/custom.py` for anything else (for
413
+ example an existing application's session cookie). Implement `authenticate` (return a `Principal` with a
414
+ stable `id` and its `roles`; 401 when the credential is missing or invalid, 503 when the issuer is
415
+ unreachable), `authorize`, and optionally `startup_problems()`, then set `auth_policy_implemented: true` in
416
+ the manifest (`deploy --env staging|prod` refuses until then).
417
+
418
+ Thread and A2A task ownership is enforced per principal. Roles in `AUTH_READ_ACROSS_ROLES` may read, never
419
+ continue or delete, other principals' threads (and list their approvals, never decide them); roles in
420
+ `AUTH_ADMIN_ROLES` manage assistants, crons and the store under langgraph-server (both empty by default).
421
+ Listing and deciding approvals are the actions `approval.read` and `approval.decide`; who may decide a
422
+ given call is then its API's `approvers`. Secrets a principal carries live only in
423
+ `attributes["credentials"]` and are never persisted, logged or traced.
424
+
425
+ ## Tracing
426
+
427
+ Off unless `TRACING_ENABLED=true`. With `LANGSMITH_API_KEY` traces go to LangSmith (`LANGSMITH_PROJECT`,
428
+ `LANGSMITH_ENDPOINT`); otherwise over OTLP/HTTP to `OTEL_EXPORTER_OTLP_ENDPOINT`. `TRACE_CAPTURE=metadata`
429
+ (default) records structure, timing, token counts, tool names, error types and hashed identifiers only;
430
+ `TRACE_CAPTURE=full` adds prompts, completions, tool arguments and results, error messages and the client's
431
+ `/chat` metadata. Run records (`runs` table under postgres, in-process under memory) follow the same policy;
432
+ client metadata is kept in the run record, never in checkpoints.
433
+ {%- if cookiecutter.deployment_target == 'kubernetes' %}
434
+
435
+ ## Environments
436
+
437
+ | Environment | Namespace | Values | Postgres |
438
+ |---|---|---|---|
439
+ | `dev` | `{{cookiecutter.project_name}}-dev` | `values.yaml` + `values-dev.yaml` | bundled subchart (`postgresql.enabled=true`) |
440
+ | `staging` | `{{cookiecutter.project_name}}-staging` | `values.yaml` + `values-staging.yaml` | external, `POSTGRES_DSN`{% if cookiecutter.runtime == 'langgraph-server' %} (`DATABASE_URI` + `REDIS_URI`){% endif %} from the Secret |
441
+ | `prod` | `{{cookiecutter.project_name}}-prod` | `values.yaml` + `values-prod.yaml` | external, from the Secret |
442
+
443
+ The Helm release name is `{{cookiecutter.project_name}}`. Contexts and namespaces are recorded under `environments:`
444
+ in `graph-agents-cli-manifest.yaml`. Traffic enters through a Gateway API `HTTPRoute` (set `gateway.parentRef` and
445
+ `gateway.hostname` in the values file) or an `Ingress` (`ingress.enabled=true`); only `route.publicPaths`
446
+ (`/chat`, `/threads`, `/a2a/{{cookiecutter.agent_directory}}`) are published, so `/health`, `/ready` and
447
+ `/metrics` stay inside the cluster. TLS comes from `tls.existingSecret` or cert-manager
448
+ (`tls.certManager.enabled=true`). The pod runs as uid 1000 with a read-only root filesystem; readiness uses
449
+ `/ready`. `image.tag` is set per deploy (the chart refuses an empty or unquoted numeric tag). Nothing is
450
+ installed by the CLI or the chart: `graph-agents-cli infra check --env <env>` reports what the cluster has.
451
+
452
+ `deploy` and `secrets apply` follow these rules:
453
+
454
+ - The env file is `--env-file`, else `.env.<env>`; only `dev` falls back to `.env`.
455
+ - The kube context is `--context`, else `environments.<env>.context`, else the kubeconfig's current one, which
456
+ outside `dev` needs a confirmation (or `--yes`). Record the staging and prod contexts in the manifest.
457
+ - Direct mode checks that the Secret holds every required key before it builds anything (exit 1 otherwise),
458
+ runs `helm upgrade --install --wait --timeout 5m`, and on a failed rollout prints the pods' states, events
459
+ and logs, then rolls back its own revision (`--atomic`, default).
460
+ - A workstation build of a tree with uncommitted changes is tagged `<sha>-dirty-<time>`.
461
+
462
+ ## Secrets
463
+
464
+ The chart never templates the app Secret; it mounts `<release>-app` (`existingSecret`) with `envFrom`, and
465
+ outside dev the pods do not start without it. Only the keys listed under `secrets.keys` in
466
+ `graph-agents-cli-manifest.yaml` are exported from an env file: the provider key, the database settings, the
467
+ policy's own keys and the token of every `auth: bearer` API in `api-policy.yaml` (`api add` and `api remove`
468
+ keep that list in step). `graph-agents-cli secrets status --env <env>` shows which of them the Secret holds.
469
+ Add other secrets you use (`METRICS_TOKEN`, `PRINCIPAL_HASH_SALT`) to that list.
470
+
471
+ ```bash
472
+ graph-agents-cli secrets apply --env staging # from .env.staging; creates the namespace if needed
473
+ graph-agents-cli secrets status --env staging # which keys are present (no values); exit 1 on a missing required key
474
+ graph-agents-cli deploy --env staging --restart # roll the pods after a rotation
475
+ ```
476
+
477
+ Secrets are applied with server-side apply; allow-listed keys the env file leaves out are kept.
478
+ {%- if cookiecutter.auth_policy == 'shared-bearer' %} The live
479
+ `API_KEY` wins: it changes only when the env file sets another one and `--rotate-api-key` is passed. A missing
480
+ `API_KEY` is generated and written to the env file (mode 0600), never printed.
481
+ {%- endif %} In `argocd` and `helm-push`
482
+ modes CI never holds application secrets: the owner named in the manifest (`secrets.owner`) runs
483
+ `secrets apply` from a workstation with cluster access, once per environment.
484
+
485
+ ## Deploying (`cd: {{cookiecutter.cd}}`)
486
+ {%- set registry_placeholder = 'CHANGE-ME' in cookiecutter.registry %}
487
+ {%- set registry = '<registry>' if registry_placeholder else cookiecutter.registry %}
488
+ {%- if registry_placeholder %}
489
+
490
+ The registry is still the placeholder `{{cookiecutter.registry}}` (`create` had no `--registry` and found no git
491
+ `origin` remote). `build` and `deploy` refuse it until you run
492
+ `graph-agents-cli scaffold enhance --registry <host>/<org>`, which sets `create_params.registry` in the manifest,
493
+ `image.repository` in the chart's `values.yaml` and `IMAGE_REPOSITORY` in `.github/agent.env`.
494
+ {%- if cookiecutter.cd != 'argocd' %} A local cluster
495
+ gets the image side-loaded, so there any valid name works (for example `--registry localhost/dev`).
496
+ {%- endif %}
497
+ {%- endif %}
498
+ {%- if cookiecutter.cd == 'skip' %}
499
+
500
+ Direct mode. `graph-agents-cli deploy --env dev` builds the image, loads it into a local cluster (kind, k3d, k3s,
501
+ minikube; nothing for Docker Desktop) or pushes it to `{{registry}}` for a remote cluster, applies the
502
+ Secret from the allow-listed keys and runs `helm upgrade --install`. `deploy --env staging|prod` works the same way
503
+ from a workstation, with the context rules above. Add CD later with
504
+ `graph-agents-cli scaffold enhance --cd argocd|helm-push`.
505
+ {%- elif cookiecutter.cd == 'argocd' %}
506
+
507
+ Pull-based. On every push to `main` the `staging` workflow builds and pushes
508
+ `{{registry}}/{{cookiecutter.project_name}}:<short sha>` (the tag a workstation `deploy` also uses),
509
+ writes the tag into `values-staging.yaml` on a branch built on the latest `main` (closing older staging PRs it
510
+ supersedes) and opens a PR with auto-merge; Argo CD reconciles `main` into `{{cookiecutter.project_name}}-staging`.
511
+ Production changes only through a PR that touches `values-prod.yaml`: the `promote-to-prod` workflow (GitHub
512
+ `production` environment) or a workstation `graph-agents-cli deploy --env prod --image <ref>` opens it, and its
513
+ merge (code-owner review, no self-approval, `pr_checks` green) is the single production gate. `deploy` never
514
+ runs helm in this mode; `deploy --status` and `--restart` and the `secrets` commands are the only cluster
515
+ operations. The production `Application` has no automated sync: sync it in Argo after the merge.
516
+ Point `deployment/argocd/application-*.yaml` at this repository (`repoURL`) and apply them to the Argo CD namespace.
517
+ {%- elif cookiecutter.cd == 'helm-push' %}
518
+
519
+ Push-based. On every push to `main` the `staging` workflow builds and pushes
520
+ `{{registry}}/{{cookiecutter.project_name}}:<short sha>` (the tag a workstation `deploy` also uses), and a
521
+ **self-hosted runner** inside the network runs `graph-agents-cli deploy --env staging --image <ref> --context <ctx> --yes`
522
+ with the kubeconfig from the `DEPLOY_KUBECONFIG` secret of the `staging` environment, then verifies the rollout
523
+ (`/health`, `/ready`). `promote-to-prod` does the same for production with the `production` environment's secret
524
+ and reviewers. From a workstation `deploy` is allowed for `dev` and refused for `staging`/`prod` (even with
525
+ `--image`) unless `--force-direct`.
526
+ {%- endif %}
527
+
528
+ ## GitHub settings the workflows rely on (not created by the CLI)
529
+
530
+ These are repository settings a workflow cannot create with the default token; `graph-agents-cli infra check`
531
+ reports whether they exist when `gh` is logged in (or `GITHUB_TOKEN` is set).
532
+
533
+ - Environment `production`: required reviewers (at least one), "prevent self-review" enabled, deployment
534
+ branches restricted to `main`, optional wait timer.
535
+ - Environment `staging`: deployment branches restricted to `main`; no reviewers.
536
+ - Branch protection on `main` (`argocd` and `helm-push`): pull requests required; required review from code
537
+ owners; "dismiss stale approvals" and "prevent self-approval" enabled; `pr_checks` as a required status
538
+ check; auto-merge allowed for the staging PR.
539
+ - `.github/CODEOWNERS` owns the chart and prod values, the workflows, `api-policy.yaml`, `tests/eval/`, the
540
+ extensions and the manifest; replace the `@CHANGE-ME/production-approvers` placeholder.
541
+ - `GH_PR_TOKEN` (a fine-grained PAT or GitHub App token with pull-request and contents write access): pull
542
+ requests opened with the workflow token never trigger `pr_checks`.
543
+ - Registry credentials: GHCR works with the workflow token; other registries take `REGISTRY_USERNAME` /
544
+ `REGISTRY_PASSWORD` secrets.
545
+ - `helm-push`: the `DEPLOY_KUBECONFIG` secret in each of the `staging` and `production` environments (never a
546
+ repository secret) and a self-hosted runner with `kubectl` and `curl`.
547
+ - Optional CI model access: the provider key as a repository secret (or the `MODEL_PROVIDER` / `MODEL_NAME`
548
+ repository variables) makes the `pr_checks` eval gate use a real model; tests always run on the `fake` one,
549
+ and on the fake model the gate warns that it is not a quality signal.
550
+ {%- endif %}
551
+
552
+ ## Evals
553
+
554
+ Datasets are JSON files in `tests/eval/datasets/` (`{"cases": [{"id", "messages", "expect", "judge", ...}]}`).
555
+ Deterministic `expect` checks and mandatory judge metrics must always pass; only the metrics listed under
556
+ `quality_metrics` in `tests/eval/eval_config.yaml` may pass at a rate below 100 percent. `eval run` exits 0 when
557
+ the gate is met, 1 on a failure, 2 on an error or missing case, 3 on a configuration error.
558
+
559
+ - `expect.contains` and `not_contains` ignore case (`case_insensitive: false` for exact case); on a multi-turn case
560
+ the checks read the final turn unless `scope: all_turns`, and the judges see every earlier turn (replies and
561
+ tool results included).
562
+ - A gate met on `MODEL_PROVIDER=fake` (or a fake judge) proves the plumbing only, and `eval grade` says so: run it
563
+ on the real provider before trusting it.
564
+ - `eval run --url <agent>` sends every case to that agent, whose tools run for real there, writes included: point
565
+ it at an environment whose data you can reset, with a test identity (`GRAPH_AGENTS_CLI_API_KEY`).
566
+
567
+ ## Coding agents
568
+
569
+ `{{cookiecutter.agent_guidance_filename}}` is the guide a coding agent reads; its `process:` line names the governing
570
+ process document ({{ cookiecutter.process if cookiecutter.process else 'none declared' }}).
571
+ Install the graph-agents-cli skills with `graph-agents-cli setup`.
@@ -0,0 +1,60 @@
1
+ # Outbound API access policy for {{cookiecutter.project_name}}.
2
+ #
3
+ # Declares every external API the agent's tools may call, and how. Owned by the
4
+ # project: `graph-agents-cli scaffold upgrade` and `enhance` never overwrite it.
5
+ # Change it with `graph-agents-cli api ...` (add, access, allow, deny, revoke,
6
+ # limits, remove; each prints the diff first) or by hand, in a reviewed pull
7
+ # request: CODEOWNERS covers this file. Widening access (more methods, more
8
+ # operations, a lifted denial) needs that review; narrowing it is always safe.
9
+ # `{{cookiecutter.agent_directory}}/app_utils/api_client.py` loads it (path from
10
+ # API_POLICY_PATH, default ./api-policy.yaml) and refuses, before sending, any
11
+ # request outside it; without this file every outbound call is refused.
12
+ # `graph-agents-cli lint` (or `graph-agents-cli api check`) checks every tool's
13
+ # API_CALLS against it with the same rules. Unknown and repeated keys are errors
14
+ # at every level, so a typo never widens access.
15
+ #
16
+ # The sample below declares one read-write API with an explicit allow-list and a
17
+ # denial. There is no default access level: choose each API's methods and
18
+ # operations (`graph-agents-cli api add --access read-only|read-write|custom`).
19
+ apis:
20
+ orders: # [a-z][a-z0-9_]*, at most 32 characters
21
+ base_url_env: ORDERS_API_BASE_URL # required; the URL may carry a path prefix
22
+ auth: bearer # required: none | bearer | forward
23
+ token_env: ORDERS_API_TOKEN # required iff auth: bearer (joins secrets.keys)
24
+ # forward_header: Authorization # auth: forward only: sends the caller's
25
+ # # attributes["credentials"]["orders"]
26
+ allowed_methods: [GET, HEAD, POST, PUT, PATCH, DELETE] # required, explicit; ["*"] = every method
27
+ allowed_operations: # optional; omit = every operation within
28
+ - operationId: listOrders # allowed_methods. An allowed entry needs every
29
+ path: /orders # field it pins (operationId, path, methods).
30
+ methods: [GET]
31
+ - operationId: getOrder
32
+ path: /orders/{order_id}
33
+ methods: [GET]
34
+ - operationId: createOrder
35
+ path: /orders
36
+ methods: [POST]
37
+ - operationId: updateOrder
38
+ path: /orders/{order_id}
39
+ methods: [PATCH]
40
+ denied_operations: # same entry shape; denials win and hold on the
41
+ - operationId: deleteOrder # path: DELETE /orders/{order_id} is refused
42
+ path: /orders/{order_id} # whatever operation_id a call gives it (a denial
43
+ methods: [DELETE] # by operationId alone knows only that label).
44
+ # openapi: docs/orders-openapi.yaml # optional; lint validates declared calls against it
45
+ timeouts_ms: {connect: 2000, read: 5000}
46
+ pagination: {page_size_param: pageSize, max_page_size: 200} # enforced at runtime
47
+ limits: {max_calls_per_run: 20, rate_per_minute: 120} # optional; per run / per replica
48
+ # approval: # optional: a person approves these calls before
49
+ # required_for: # they are sent (README "Human approval of calls");
50
+ # methods: [POST, PATCH] # a gated call must still be allowed above
51
+ # operations: # and/or entries shaped like allowed_operations
52
+ # - operationId: updateOrder # (pin path and methods too: an entry by
53
+ # path: /orders/{order_id} # operationId alone gates every call that
54
+ # methods: [PATCH] # names no operation_id)
55
+ # approvers: [requester] # "requester" and/or "role:<name>" (four eyes)
56
+ # timeout_s: 900 # 30..86400; unanswered in time = rejected
57
+ # (or a list of such rules, for other approvers on other calls: the first rule
58
+ # that covers a call gates it; `graph-agents-cli api approval --add-rule`. A call
59
+ # an earlier rule covers only because it names no operation_id, and that a later
60
+ # rule with other approvers also covers, is refused.)