graph-agents-cli 0.3.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (291) hide show
  1. graph_agents_cli/__init__.py +26 -0
  2. graph_agents_cli/_api_policy.py +2145 -0
  3. graph_agents_cli/_approvals.py +400 -0
  4. graph_agents_cli/_build.py +186 -0
  5. graph_agents_cli/_build_info.json +7 -0
  6. graph_agents_cli/_chat_client.py +462 -0
  7. graph_agents_cli/_click.py +157 -0
  8. graph_agents_cli/_defaults.py +139 -0
  9. graph_agents_cli/_experiments.py +64 -0
  10. graph_agents_cli/_http.py +192 -0
  11. graph_agents_cli/_output.py +83 -0
  12. graph_agents_cli/_project.py +462 -0
  13. graph_agents_cli/_remote.py +220 -0
  14. graph_agents_cli/_response_schema.py +264 -0
  15. graph_agents_cli/_runner.py +319 -0
  16. graph_agents_cli/_skills_check.py +274 -0
  17. graph_agents_cli/_tools.py +189 -0
  18. graph_agents_cli/_trust.py +66 -0
  19. graph_agents_cli/api/__init__.py +15 -0
  20. graph_agents_cli/api/_changes.py +506 -0
  21. graph_agents_cli/api/_files.py +658 -0
  22. graph_agents_cli/api/cmd_api.py +2480 -0
  23. graph_agents_cli/deploy/__init__.py +15 -0
  24. graph_agents_cli/deploy/_config.py +171 -0
  25. graph_agents_cli/deploy/_image.py +128 -0
  26. graph_agents_cli/deploy/_kube.py +286 -0
  27. graph_agents_cli/deploy/_modes.py +234 -0
  28. graph_agents_cli/deploy/_preflight.py +370 -0
  29. graph_agents_cli/deploy/_values.py +168 -0
  30. graph_agents_cli/deploy/cmd_deploy.py +1866 -0
  31. graph_agents_cli/deploy/gitops.py +562 -0
  32. graph_agents_cli/deploy/local_load.py +273 -0
  33. graph_agents_cli/dev/__init__.py +13 -0
  34. graph_agents_cli/dev/cmd_build.py +131 -0
  35. graph_agents_cli/dev/cmd_install.py +78 -0
  36. graph_agents_cli/dev/cmd_lint.py +119 -0
  37. graph_agents_cli/dev/cmd_playground.py +297 -0
  38. graph_agents_cli/dev/policy_check.py +1287 -0
  39. graph_agents_cli/eval/__init__.py +22 -0
  40. graph_agents_cli/eval/_client.py +670 -0
  41. graph_agents_cli/eval/_common.py +177 -0
  42. graph_agents_cli/eval/_judge.py +168 -0
  43. graph_agents_cli/eval/_judge_runner.py +238 -0
  44. graph_agents_cli/eval/_paths.py +212 -0
  45. graph_agents_cli/eval/checks.py +581 -0
  46. graph_agents_cli/eval/cmd_analyze.py +278 -0
  47. graph_agents_cli/eval/cmd_compare.py +284 -0
  48. graph_agents_cli/eval/cmd_eval_group.py +80 -0
  49. graph_agents_cli/eval/cmd_generate.py +558 -0
  50. graph_agents_cli/eval/cmd_grade.py +466 -0
  51. graph_agents_cli/eval/cmd_metric.py +156 -0
  52. graph_agents_cli/eval/cmd_run.py +370 -0
  53. graph_agents_cli/eval/cmd_submit.py +400 -0
  54. graph_agents_cli/eval/config.py +435 -0
  55. graph_agents_cli/eval/dataset.py +350 -0
  56. graph_agents_cli/eval/gate.py +420 -0
  57. graph_agents_cli/eval/transcript.py +192 -0
  58. graph_agents_cli/extension/__init__.py +13 -0
  59. graph_agents_cli/extension/_compat.py +86 -0
  60. graph_agents_cli/extension/_loader.py +293 -0
  61. graph_agents_cli/extension/_manifest.py +135 -0
  62. graph_agents_cli/extension/_overrides.py +195 -0
  63. graph_agents_cli/extension/_paths.py +91 -0
  64. graph_agents_cli/extension/_refs.py +193 -0
  65. graph_agents_cli/extension/_resolver.py +453 -0
  66. graph_agents_cli/extension/_schema.py +106 -0
  67. graph_agents_cli/extension/_spec.py +253 -0
  68. graph_agents_cli/extension/_sync.py +102 -0
  69. graph_agents_cli/extension/_trust.py +58 -0
  70. graph_agents_cli/extension/cmd_extension_add.py +259 -0
  71. graph_agents_cli/extension/cmd_extension_group.py +57 -0
  72. graph_agents_cli/extension/cmd_extension_list.py +56 -0
  73. graph_agents_cli/extension/cmd_extension_remove.py +61 -0
  74. graph_agents_cli/extension/cmd_extension_update.py +195 -0
  75. graph_agents_cli/info/__init__.py +13 -0
  76. graph_agents_cli/info/cmd_info.py +222 -0
  77. graph_agents_cli/infra/__init__.py +15 -0
  78. graph_agents_cli/infra/checks.py +1169 -0
  79. graph_agents_cli/infra/cmd_infra.py +103 -0
  80. graph_agents_cli/main.py +591 -0
  81. graph_agents_cli/peer/__init__.py +15 -0
  82. graph_agents_cli/peer/_generate.py +254 -0
  83. graph_agents_cli/peer/cmd_peer.py +1151 -0
  84. graph_agents_cli/run/__init__.py +13 -0
  85. graph_agents_cli/run/_local_server.py +1157 -0
  86. graph_agents_cli/run/_signals.py +141 -0
  87. graph_agents_cli/run/cmd_approvals.py +530 -0
  88. graph_agents_cli/run/cmd_run.py +1421 -0
  89. graph_agents_cli/scaffold/__init__.py +19 -0
  90. graph_agents_cli/scaffold/agents/README.md +24 -0
  91. graph_agents_cli/scaffold/agents/empty_py/.template/templateconfig.yaml +22 -0
  92. graph_agents_cli/scaffold/agents/langgraph/.env.example +292 -0
  93. graph_agents_cli/scaffold/agents/langgraph/.template/templateconfig.yaml +28 -0
  94. graph_agents_cli/scaffold/agents/langgraph/Dockerfile +59 -0
  95. graph_agents_cli/scaffold/agents/langgraph/Dockerfile.langgraph-server +59 -0
  96. graph_agents_cli/scaffold/agents/langgraph/README.md +571 -0
  97. graph_agents_cli/scaffold/agents/langgraph/api-policy.yaml +60 -0
  98. graph_agents_cli/scaffold/agents/langgraph/app/__init__.py +20 -0
  99. graph_agents_cli/scaffold/agents/langgraph/app/agent.py +174 -0
  100. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/__init__.py +15 -0
  101. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/a2a.py +2162 -0
  102. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/a2a_client.py +1167 -0
  103. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/api_client.py +4220 -0
  104. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/approvals.py +1349 -0
  105. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/auth.py +1986 -0
  106. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/chat.py +2962 -0
  107. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/checkpointer.py +432 -0
  108. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/content.py +569 -0
  109. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/db.py +580 -0
  110. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/limits.py +203 -0
  111. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/metrics.py +231 -0
  112. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/middleware.py +361 -0
  113. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/model.py +611 -0
  114. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/playground.py +230 -0
  115. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/run_locks.py +459 -0
  116. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/structured.py +755 -0
  117. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/telemetry.py +681 -0
  118. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/threads.py +493 -0
  119. graph_agents_cli/scaffold/agents/langgraph/app/app_utils/token_exchange.py +959 -0
  120. graph_agents_cli/scaffold/agents/langgraph/app/fast_api_app.py +770 -0
  121. graph_agents_cli/scaffold/agents/langgraph/app/policies/__init__.py +55 -0
  122. graph_agents_cli/scaffold/agents/langgraph/app/policies/custom.py +97 -0
  123. graph_agents_cli/scaffold/agents/langgraph/app/tools/__init__.py +46 -0
  124. graph_agents_cli/scaffold/agents/langgraph/app/tools/example_api.py +92 -0
  125. graph_agents_cli/scaffold/agents/langgraph/app/tools/weather.py +33 -0
  126. graph_agents_cli/scaffold/agents/langgraph/langgraph.json +14 -0
  127. graph_agents_cli/scaffold/agents/langgraph/pyproject.toml +78 -0
  128. graph_agents_cli/scaffold/agents/langgraph/tests/conftest.py +376 -0
  129. graph_agents_cli/scaffold/agents/langgraph/tests/eval/datasets/basic-dataset.json +53 -0
  130. graph_agents_cli/scaffold/agents/langgraph/tests/eval/eval_config.yaml +32 -0
  131. graph_agents_cli/scaffold/agents/langgraph/tests/integration/approval_graph.py +137 -0
  132. graph_agents_cli/scaffold/agents/langgraph/tests/integration/fake_issuer.py +216 -0
  133. graph_agents_cli/scaffold/agents/langgraph/tests/integration/fake_openai.py +357 -0
  134. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_a2a_outcomes.py +569 -0
  135. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_a2a_relay.py +479 -0
  136. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_api_surface.py +812 -0
  137. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_approvals.py +1367 -0
  138. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_approvals_server.py +794 -0
  139. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_cross_actor.py +497 -0
  140. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_cross_actor_server.py +247 -0
  141. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_history_repair.py +278 -0
  142. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_model_apis.py +242 -0
  143. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_postgres.py +637 -0
  144. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_resilience_postgres.py +770 -0
  145. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_runtime_guardrails.py +854 -0
  146. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_server_e2e.py +340 -0
  147. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_server_runtime.py +989 -0
  148. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_structured_answers.py +584 -0
  149. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_structured_server.py +222 -0
  150. graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_token_exchange_issuer.py +650 -0
  151. graph_agents_cli/scaffold/agents/langgraph/tests/load_test/.results/.placeholder +0 -0
  152. graph_agents_cli/scaffold/agents/langgraph/tests/load_test/README.md +22 -0
  153. graph_agents_cli/scaffold/agents/langgraph/tests/load_test/conftest.py +21 -0
  154. graph_agents_cli/scaffold/agents/langgraph/tests/load_test/load_test.py +81 -0
  155. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_a2a_client.py +824 -0
  156. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_a2a_scoping.py +724 -0
  157. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_api_client.py +1214 -0
  158. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_api_client_hardening.py +716 -0
  159. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_api_policy_rpc.py +767 -0
  160. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_approval_ledger.py +1536 -0
  161. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_fake_model.py +115 -0
  162. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_jwt_policy.py +991 -0
  163. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_limits.py +310 -0
  164. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_logging.py +148 -0
  165. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_logging_hardening.py +271 -0
  166. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_policy.py +378 -0
  167. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_resilience.py +610 -0
  168. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_server_auth.py +702 -0
  169. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_structured.py +673 -0
  170. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_telemetry.py +404 -0
  171. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_thread_listing.py +255 -0
  172. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_threads.py +268 -0
  173. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_token_exchange.py +1320 -0
  174. graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_untrusted_content.py +393 -0
  175. graph_agents_cli/scaffold/agents/langgraph/uv-fastapi.lock +2084 -0
  176. graph_agents_cli/scaffold/agents/langgraph/uv-langgraph-server.lock +2106 -0
  177. graph_agents_cli/scaffold/agents/langgraph/{{cookiecutter.agent_guidance_filename}} +129 -0
  178. graph_agents_cli/scaffold/base_templates/_shared/graph-agents-cli-manifest.yaml +36 -0
  179. graph_agents_cli/scaffold/base_templates/python/.dockerignore +32 -0
  180. graph_agents_cli/scaffold/base_templates/python/.github/CODEOWNERS +30 -0
  181. graph_agents_cli/scaffold/base_templates/python/.github/agent.env +7 -0
  182. graph_agents_cli/scaffold/base_templates/python/.github/workflows/pr_checks.yaml +214 -0
  183. graph_agents_cli/scaffold/base_templates/python/.gitignore +209 -0
  184. graph_agents_cli/scaffold/base_templates/python/tests/unit/test_dummy.py +23 -0
  185. graph_agents_cli/scaffold/base_templates/python/{{cookiecutter.agent_guidance_filename}} +35 -0
  186. graph_agents_cli/scaffold/cmd_scaffold_group.py +49 -0
  187. graph_agents_cli/scaffold/commands/__init__.py +13 -0
  188. graph_agents_cli/scaffold/commands/create.py +1424 -0
  189. graph_agents_cli/scaffold/commands/enhance.py +1652 -0
  190. graph_agents_cli/scaffold/commands/upgrade.py +570 -0
  191. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/.github/agent.env +12 -0
  192. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/.github/workflows/promote-to-prod.yaml +371 -0
  193. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/.github/workflows/staging.yaml +450 -0
  194. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/argocd/application-dev.yaml +43 -0
  195. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/argocd/application-prod.yaml +41 -0
  196. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/argocd/application-staging.yaml +43 -0
  197. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/.helmignore +14 -0
  198. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/Chart.yaml +21 -0
  199. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/examples/networkpolicy.yaml +103 -0
  200. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/NOTES.txt +48 -0
  201. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/_helpers.tpl +189 -0
  202. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/certificate.yaml +15 -0
  203. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/configmap.yaml +10 -0
  204. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/deployment.yaml +199 -0
  205. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/hpa.yaml +22 -0
  206. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/httproute.yaml +30 -0
  207. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/ingress.yaml +39 -0
  208. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/networkpolicy.yaml +48 -0
  209. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/pdb.yaml +13 -0
  210. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/postgresql-secret.yaml +37 -0
  211. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/service.yaml +15 -0
  212. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/serviceaccount.yaml +13 -0
  213. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/servicemonitor.yaml +42 -0
  214. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values-dev.yaml +22 -0
  215. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values-prod.yaml +45 -0
  216. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values-staging.yaml +29 -0
  217. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values.yaml +396 -0
  218. graph_agents_cli/scaffold/deployment_targets/kubernetes/python/tests/integration/test_chart.py +269 -0
  219. graph_agents_cli/scaffold/deployment_targets/none/README.md +5 -0
  220. graph_agents_cli/scaffold/deployment_targets/none/python/README.md +6 -0
  221. graph_agents_cli/scaffold/utils/__init__.py +13 -0
  222. graph_agents_cli/scaffold/utils/backup.py +212 -0
  223. graph_agents_cli/scaffold/utils/build_record.py +257 -0
  224. graph_agents_cli/scaffold/utils/cli_options.py +184 -0
  225. graph_agents_cli/scaffold/utils/fs.py +83 -0
  226. graph_agents_cli/scaffold/utils/generate_locks.py +214 -0
  227. graph_agents_cli/scaffold/utils/generation_metadata.py +88 -0
  228. graph_agents_cli/scaffold/utils/keyedit.py +768 -0
  229. graph_agents_cli/scaffold/utils/keymerge.py +537 -0
  230. graph_agents_cli/scaffold/utils/language.py +138 -0
  231. graph_agents_cli/scaffold/utils/lock_utils.py +94 -0
  232. graph_agents_cli/scaffold/utils/logging.py +77 -0
  233. graph_agents_cli/scaffold/utils/manifest.py +292 -0
  234. graph_agents_cli/scaffold/utils/merge.py +970 -0
  235. graph_agents_cli/scaffold/utils/merge3.py +216 -0
  236. graph_agents_cli/scaffold/utils/openapi_seed.py +199 -0
  237. graph_agents_cli/scaffold/utils/remote_template.py +376 -0
  238. graph_agents_cli/scaffold/utils/template.py +1352 -0
  239. graph_agents_cli/scaffold/utils/upgrade.py +894 -0
  240. graph_agents_cli/scaffold/utils/version.py +438 -0
  241. graph_agents_cli/secrets/__init__.py +15 -0
  242. graph_agents_cli/secrets/_apply.py +954 -0
  243. graph_agents_cli/secrets/_required.py +188 -0
  244. graph_agents_cli/secrets/cmd_secrets.py +211 -0
  245. graph_agents_cli/setup/__init__.py +13 -0
  246. graph_agents_cli/setup/_antigravity.py +221 -0
  247. graph_agents_cli/setup/cmd_auth.py +1030 -0
  248. graph_agents_cli/setup/cmd_dev_token.py +513 -0
  249. graph_agents_cli/setup/cmd_setup.py +428 -0
  250. graph_agents_cli/setup/cmd_update.py +140 -0
  251. graph_agents_cli/skills/__init__.py +13 -0
  252. graph_agents_cli/skills/_bundle.py +65 -0
  253. graph_agents_cli/skills/data/README.md +19 -0
  254. graph_agents_cli/skills/data/graph-agents-cli-deploy/SKILL.md +357 -0
  255. graph_agents_cli/skills/data/graph-agents-cli-deploy/references/github-settings.md +113 -0
  256. graph_agents_cli/skills/data/graph-agents-cli-deploy/references/gitops.md +137 -0
  257. graph_agents_cli/skills/data/graph-agents-cli-deploy/references/kubernetes.md +315 -0
  258. graph_agents_cli/skills/data/graph-agents-cli-deploy/references/secrets.md +160 -0
  259. graph_agents_cli/skills/data/graph-agents-cli-eval/SKILL.md +303 -0
  260. graph_agents_cli/skills/data/graph-agents-cli-eval/references/dataset_schema.md +282 -0
  261. graph_agents_cli/skills/data/graph-agents-cli-eval/references/metrics-guide.md +143 -0
  262. graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/SKILL.md +659 -0
  263. graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/references/langchain-models.md +124 -0
  264. graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/references/langgraph.md +235 -0
  265. graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/references/template-contract.md +477 -0
  266. graph_agents_cli/skills/data/graph-agents-cli-observability/SKILL.md +231 -0
  267. graph_agents_cli/skills/data/graph-agents-cli-observability/references/langsmith.md +46 -0
  268. graph_agents_cli/skills/data/graph-agents-cli-observability/references/otel.md +59 -0
  269. graph_agents_cli/skills/data/graph-agents-cli-scaffold/SKILL.md +414 -0
  270. graph_agents_cli/skills/data/graph-agents-cli-scaffold/references/flags.md +134 -0
  271. graph_agents_cli/skills/data/graph-agents-cli-workflow/SKILL.md +478 -0
  272. graph_agents_cli/skills/data/graph-agents-cli-workflow/references/brainstorming.md +118 -0
  273. graph_agents_cli/skills/data/graph-agents-cli-workflow/references/commands.md +419 -0
  274. graph_agents_cli/skills/data/graph-agents-cli-workflow/references/extension.md +156 -0
  275. graph_agents_cli/skills/data/graph-agents-cli-workflow/references/internals.md +67 -0
  276. graph_agents_cli/skills/data/graph-agents-cli-workflow/references/spec-template.md +56 -0
  277. graph_agents_cli/skills/data/graph-agents-cli-workflow/references/terminology.md +119 -0
  278. graph_agents_cli/system/__init__.py +15 -0
  279. graph_agents_cli/system/_apply.py +519 -0
  280. graph_agents_cli/system/_checks.py +1023 -0
  281. graph_agents_cli/system/_deploy.py +215 -0
  282. graph_agents_cli/system/_model.py +363 -0
  283. graph_agents_cli/system/_system.py +664 -0
  284. graph_agents_cli/system/_views.py +208 -0
  285. graph_agents_cli/system/cmd_system.py +423 -0
  286. graph_agents_cli-0.3.1.dist-info/METADATA +162 -0
  287. graph_agents_cli-0.3.1.dist-info/RECORD +291 -0
  288. graph_agents_cli-0.3.1.dist-info/WHEEL +4 -0
  289. graph_agents_cli-0.3.1.dist-info/entry_points.txt +2 -0
  290. graph_agents_cli-0.3.1.dist-info/licenses/LICENSE +201 -0
  291. graph_agents_cli-0.3.1.dist-info/licenses/NOTICE +19 -0
@@ -0,0 +1,770 @@
1
+ # Copyright 2026 graph-agents-cli contributors
2
+ #
3
+ # Licensed under the Apache License, Version 2.0 (the "License");
4
+ # you may not use this file except in compliance with the License.
5
+ # You may obtain a copy of the License at
6
+ #
7
+ # https://www.apache.org/licenses/LICENSE-2.0
8
+ #
9
+ # Unless required by applicable law or agreed to in writing, software
10
+ # distributed under the License is distributed on an "AS IS" BASIS,
11
+ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
12
+ # See the License for the specific language governing permissions and
13
+ # limitations under the License.
14
+
15
+ """Failures against a real Postgres: lost replicas, dropped sessions, outages, crashes,
16
+ and A2A tasks across replicas.
17
+
18
+ Opt-in: set `TEST_POSTGRES_DSN` to the URL of a server where the user may
19
+ create databases (each test gets a fresh one, dropped afterwards). Outages are
20
+ made with a TCP proxy in front of that server, which the tests stop, restart
21
+ and cut; crashes with a real server process killed with SIGKILL.
22
+ """
23
+
24
+ from __future__ import annotations
25
+
26
+ import asyncio
27
+ import contextlib
28
+ import json
29
+ import logging
30
+ import os
31
+ import signal
32
+ import socket
33
+ import subprocess
34
+ import sys
35
+ import time
36
+ import uuid
37
+ from collections.abc import AsyncIterator
38
+ from pathlib import Path
39
+ from typing import Any
40
+ from urllib.parse import urlsplit
41
+
42
+ os.environ.update(
43
+ {
44
+ "MODEL_PROVIDER": "fake",
45
+ "MODEL_NAME": "fake",
46
+ "AUTH_POLICY": "shared-bearer",
47
+ "API_KEY": "test-key",
48
+ "APP_ENV": "dev",
49
+ "TRACING_ENABLED": "false",
50
+ "RUNTIME": "fastapi",
51
+ "APP_URL": "http://testserver",
52
+ }
53
+ )
54
+
55
+ import httpx
56
+ import pytest
57
+
58
+ from {{cookiecutter.agent_directory}}.app_utils import chat as chat_module
59
+ from {{cookiecutter.agent_directory}}.app_utils import run_locks
60
+ from {{cookiecutter.agent_directory}}.app_utils.checkpointer import POSTGRES
61
+ from {{cookiecutter.agent_directory}}.app_utils.db import Database, RunRecord, RunStore
62
+ from {{cookiecutter.agent_directory}}.app_utils.threads import LeaseLost, ThreadBusy, ThreadLocks
63
+
64
+ ADMIN_DSN = os.environ.get("TEST_POSTGRES_DSN", "")
65
+ pytestmark = pytest.mark.skipif(not ADMIN_DSN, reason="TEST_POSTGRES_DSN is not set")
66
+ AUTH = {"Authorization": "Bearer test-key"}
67
+ PROJECT = Path(__file__).resolve().parents[2]
68
+ AGENT_DIR = "{{cookiecutter.agent_directory}}"
69
+ # Short leases, so a lost replica is noticed in seconds.
70
+ FAST_LEASES = {"ttl_s": 2.0, "renew_every_s": 0.2, "validity_s": 1.0}
71
+
72
+
73
+ def _with_database(dsn: str, name: str) -> str:
74
+ return urlsplit(dsn)._replace(path=f"/{name}").geturl()
75
+
76
+
77
+ def _with_port(dsn: str, port: int) -> str:
78
+ parts = urlsplit(dsn)
79
+ userinfo = parts.netloc.rsplit("@", 1)[0] + "@" if "@" in parts.netloc else ""
80
+ return parts._replace(netloc=f"{userinfo}127.0.0.1:{port}").geturl()
81
+
82
+
83
+ @pytest.fixture
84
+ async def dsn() -> AsyncIterator[str]:
85
+ import psycopg
86
+
87
+ name = f"gac_test_{uuid.uuid4().hex[:12]}"
88
+ async with await psycopg.AsyncConnection.connect(ADMIN_DSN, autocommit=True) as admin:
89
+ await admin.execute(f'CREATE DATABASE "{name}"')
90
+ yield _with_database(ADMIN_DSN, name)
91
+ async with await psycopg.AsyncConnection.connect(ADMIN_DSN, autocommit=True) as admin:
92
+ await admin.execute(f'DROP DATABASE IF EXISTS "{name}" WITH (FORCE)')
93
+
94
+
95
+ async def _schema(dsn: str) -> None:
96
+ db = Database(POSTGRES, dsn)
97
+ await db.open()
98
+ await db.close()
99
+
100
+
101
+ async def _sql(dsn: str, sql: str, params: tuple[Any, ...] = ()) -> list[tuple[Any, ...]]:
102
+ import psycopg
103
+
104
+ async with await psycopg.AsyncConnection.connect(dsn, autocommit=True) as conn:
105
+ cur = await conn.execute(sql, params)
106
+ return list(await cur.fetchall()) if cur.description else []
107
+
108
+
109
+ async def _acquire_within(locks: ThreadLocks, thread_id: str, seconds: float) -> float:
110
+ """Seconds until `locks` gets the thread (fails after `seconds`)."""
111
+ start = time.monotonic()
112
+ while True:
113
+ try:
114
+ lease = await locks.acquire(thread_id)
115
+ except ThreadBusy:
116
+ if time.monotonic() - start > seconds:
117
+ raise
118
+ await asyncio.sleep(0.1)
119
+ continue
120
+ await lease.release()
121
+ return time.monotonic() - start
122
+
123
+
124
+ # --- run leases ----------------------------------------------------------------------------
125
+
126
+
127
+ async def test_a_replica_lost_without_closing_its_connection_frees_its_threads(
128
+ dsn: str,
129
+ ) -> None:
130
+ """A frozen or partitioned replica: its session stays open, it just stops renewing."""
131
+ await _schema(dsn)
132
+ lost_replica, other = ThreadLocks(dsn, **FAST_LEASES), ThreadLocks(dsn, **FAST_LEASES)
133
+ try:
134
+ lease = await lost_replica.acquire("t1")
135
+ stopped: list[str] = []
136
+ lease.on_lost(lambda: stopped.append("stopped"))
137
+ with pytest.raises(ThreadBusy):
138
+ await other.acquire("t1")
139
+ heartbeat = lost_replica._task
140
+ assert heartbeat is not None
141
+ heartbeat.cancel() # frozen: no renewals, the connection stays open
142
+ waited = await _acquire_within(other, "t1", 10)
143
+ assert 1.0 < waited < 6.0 # the 2 s lease, not the OS's 2 h TCP keepalive
144
+ # The lost replica may no longer write the thread, and knows it.
145
+ with pytest.raises(LeaseLost):
146
+ lost_replica.fence("t1")
147
+ assert stopped == ["stopped"] and lease.lost
148
+ # Back from the partition: its release does not free what it no longer holds.
149
+ taken = await other.acquire("t1")
150
+ await lease.release()
151
+ third = ThreadLocks(dsn, **FAST_LEASES)
152
+ try:
153
+ with pytest.raises(ThreadBusy):
154
+ await third.acquire("t1")
155
+ finally:
156
+ await third.close()
157
+ await taken.release()
158
+ finally:
159
+ await lost_replica.close()
160
+ await other.close()
161
+
162
+
163
+ async def test_replicas_racing_for_a_thread_get_it_once(dsn: str) -> None:
164
+ await _schema(dsn)
165
+ replicas = [ThreadLocks(dsn, **FAST_LEASES) for _ in range(8)]
166
+ try:
167
+ results = await asyncio.gather(*(r.acquire("t1") for r in replicas), return_exceptions=True)
168
+ won = [r for r in results if not isinstance(r, BaseException)]
169
+ assert len(won) == 1
170
+ assert all(isinstance(r, ThreadBusy) for r in results if r not in won)
171
+ tokens = await _sql(dsn, "SELECT token FROM thread_locks WHERE thread_id = 't1'")
172
+ assert tokens == [(won[0].token,)]
173
+ await won[0].release()
174
+ # The next holder gets a new fencing token.
175
+ again = await replicas[3].acquire("t1")
176
+ assert again.token is not None and again.token > won[0].token
177
+ await again.release()
178
+ finally:
179
+ for r in replicas:
180
+ await r.close()
181
+
182
+
183
+ async def test_a_dropped_session_keeps_the_lease_and_its_run(dsn: str) -> None:
184
+ """What a Postgres restart or failover, an admin kill or a proxy reset does to a session."""
185
+ await _schema(dsn)
186
+ holder, other = ThreadLocks(dsn, **FAST_LEASES), ThreadLocks(dsn, **FAST_LEASES)
187
+ try:
188
+ lease = await holder.acquire("t1")
189
+ rows = await _sql(
190
+ dsn,
191
+ "SELECT count(pg_terminate_backend(pid)) FROM pg_stat_activity "
192
+ "WHERE datname = current_database() AND pid <> pg_backend_pid()",
193
+ )
194
+ assert rows[0][0] >= 1
195
+ with pytest.raises(ThreadBusy): # the old advisory lock was gone at this point
196
+ await other.acquire("t1")
197
+ await asyncio.sleep(FAST_LEASES["validity_s"] * 2) # renewals reconnect
198
+ holder.fence("t1")
199
+ assert not lease.lost
200
+ with pytest.raises(ThreadBusy):
201
+ await other.acquire("t1")
202
+ await lease.release()
203
+ assert await _acquire_within(other, "t1", 2) < 1.0
204
+ finally:
205
+ await holder.close()
206
+ await other.close()
207
+
208
+
209
+ async def test_a_lease_whose_release_failed_is_released_in_the_background(
210
+ dsn: str, monkeypatch: pytest.MonkeyPatch
211
+ ) -> None:
212
+ await _schema(dsn)
213
+ holder, other = ThreadLocks(dsn, **FAST_LEASES), ThreadLocks(dsn, **FAST_LEASES)
214
+ try:
215
+ execute = holder._execute
216
+ fail_release = {"t1", "t2"} # the first release of each fails
217
+
218
+ async def flaky(sql: str, params: dict[str, Any], **kwargs: Any) -> Any:
219
+ if sql.startswith("DELETE") and params.get("thread") in fail_release:
220
+ fail_release.discard(params["thread"])
221
+ raise OSError("connection refused")
222
+ return await execute(sql, params, **kwargs)
223
+
224
+ monkeypatch.setattr(holder, "_execute", flaky)
225
+ lease = await holder.acquire("t1")
226
+ await lease.release()
227
+ assert "t1" in holder._unreleased and "t1" not in holder.held
228
+ # Every replica gets it once the background retry went through...
229
+ assert await _acquire_within(other, "t1", 5) < 4
230
+ # ...and this process takes a leftover of its own back at once.
231
+ lease = await holder.acquire("t2")
232
+ await lease.release()
233
+ assert "t2" in holder._unreleased
234
+ again = await holder.acquire("t2")
235
+ assert "t2" not in holder._unreleased
236
+ await again.release()
237
+ assert await _acquire_within(other, "t2", 2) < 1
238
+ finally:
239
+ await holder.close()
240
+ await other.close()
241
+
242
+
243
+ async def test_runs_of_dead_processes_are_reconciled_as_interrupted(dsn: str) -> None:
244
+ await _schema(dsn)
245
+ db = Database(POSTGRES, dsn)
246
+ await db.open()
247
+ locks = ThreadLocks(dsn, **FAST_LEASES)
248
+ try:
249
+ runs = RunStore(db)
250
+ old = "2000-01-01T00:00:00+00:00"
251
+ for run_id, thread_id in (("dead", "t-dead"), ("alive", "t-alive")):
252
+ await runs.start(
253
+ RunRecord(
254
+ run_id=run_id,
255
+ thread_id=thread_id,
256
+ principal_hash="h",
257
+ model="m",
258
+ status="ok",
259
+ created_at=old,
260
+ )
261
+ )
262
+ lease = await locks.acquire("t-alive") # a run still holds its thread
263
+ replicas = await asyncio.gather(runs.reconcile(0), runs.reconcile(0))
264
+ assert sorted(replicas, key=len) == [[], ["dead"]] # closed once across replicas
265
+ dead, alive = await runs.get("dead"), await runs.get("alive")
266
+ assert dead is not None and dead.status == "interrupted"
267
+ assert dead.error_type == "ProcessLost"
268
+ assert alive is not None and alive.status == "running"
269
+ await lease.release()
270
+ assert await runs.reconcile(0) == ["alive"]
271
+ finally:
272
+ await locks.close()
273
+ await db.close()
274
+
275
+
276
+ # --- a TCP proxy that can take the database away ---------------------------------------------
277
+
278
+
279
+ class Proxy:
280
+ """Forwards 127.0.0.1:<port> to the test server; `down()` refuses and cuts every connection."""
281
+
282
+ def __init__(self, upstream: str) -> None:
283
+ parts = urlsplit(upstream)
284
+ self.host, self.upstream_port = parts.hostname or "127.0.0.1", parts.port or 5432
285
+ self.port = 0
286
+ self.server: asyncio.base_events.Server | None = None
287
+ self.writers: set[asyncio.StreamWriter] = set()
288
+
289
+ async def up(self) -> None:
290
+ self.server = await asyncio.start_server(self._handle, "127.0.0.1", self.port)
291
+ self.port = self.server.sockets[0].getsockname()[1]
292
+
293
+ async def down(self) -> None:
294
+ server, self.server = self.server, None
295
+ if server is not None:
296
+ server.close() # refuse new connections
297
+ for writer in list(self.writers): # cut the open ones
298
+ writer.close()
299
+ self.writers.clear()
300
+ if server is not None:
301
+ with contextlib.suppress(TimeoutError):
302
+ await asyncio.wait_for(server.wait_closed(), 5)
303
+
304
+ async def _handle(self, reader: asyncio.StreamReader, writer: asyncio.StreamWriter) -> None:
305
+ try:
306
+ up_reader, up_writer = await asyncio.open_connection(self.host, self.upstream_port)
307
+ except OSError:
308
+ writer.close()
309
+ return
310
+ self.writers.update((writer, up_writer))
311
+
312
+ async def pipe(src: asyncio.StreamReader, dst: asyncio.StreamWriter) -> None:
313
+ with contextlib.suppress(Exception):
314
+ while data := await src.read(65536):
315
+ dst.write(data)
316
+ await dst.drain()
317
+ dst.close()
318
+
319
+ await asyncio.gather(pipe(reader, up_writer), pipe(up_reader, writer))
320
+ self.writers.difference_update((writer, up_writer))
321
+
322
+
323
+ @pytest.fixture
324
+ async def proxy(dsn: str) -> AsyncIterator[Proxy]:
325
+ p = Proxy(dsn)
326
+ await p.up()
327
+ yield p
328
+ await p.down()
329
+
330
+
331
+ @pytest.fixture
332
+ def postgres_app(dsn: str, proxy: Proxy, monkeypatch: pytest.MonkeyPatch) -> tuple[Any, str]:
333
+ """The app configured for Postgres through the proxy (enter its lifespan in the test)."""
334
+ from {{cookiecutter.agent_directory}}.fast_api_app import app
335
+
336
+ for name in ("RETENTION_DAYS", "RECURSION_LIMIT", "RUN_TIMEOUT_S", "SSE_HEARTBEAT_S"):
337
+ monkeypatch.delenv(name, raising=False)
338
+ monkeypatch.setenv("CHECKPOINTER", "postgres")
339
+ monkeypatch.setenv("POSTGRES_DSN", _with_port(dsn, proxy.port))
340
+ monkeypatch.setattr(run_locks, "LEASE_TTL_S", FAST_LEASES["ttl_s"])
341
+ monkeypatch.setattr(run_locks, "RENEW_EVERY_S", FAST_LEASES["renew_every_s"])
342
+ monkeypatch.setattr(run_locks, "LOCAL_VALIDITY_S", FAST_LEASES["validity_s"])
343
+ monkeypatch.setattr(chat_module, "INIT_RETRY_MAX_S", 0.5)
344
+ monkeypatch.setattr(chat_module, "RECONCILE_INTERVAL_S", 0.5)
345
+ return app, dsn
346
+
347
+
348
+ def _client(app: Any) -> httpx.AsyncClient:
349
+ transport = httpx.ASGITransport(app=app, raise_app_exceptions=False)
350
+ return httpx.AsyncClient(transport=transport, base_url="http://testserver", timeout=30)
351
+
352
+
353
+ def parse_sse(text: str) -> list[tuple[str, dict[str, Any]]]:
354
+ events: list[tuple[str, dict[str, Any]]] = []
355
+ event = None
356
+ for line in text.splitlines():
357
+ if line.startswith("event:"):
358
+ event = line[6:].strip()
359
+ elif line.startswith("data:") and event:
360
+ events.append((event, json.loads(line[5:].strip())))
361
+ event = None
362
+ return events
363
+
364
+
365
+ async def _ready_within(client: httpx.AsyncClient, seconds: float) -> float:
366
+ start = time.monotonic()
367
+ while (await client.get("/ready")).status_code != 200:
368
+ assert time.monotonic() - start < seconds, "not ready in time"
369
+ await asyncio.sleep(0.2)
370
+ return time.monotonic() - start
371
+
372
+
373
+ async def test_the_app_starts_while_the_database_is_down_and_gets_ready_after_it(
374
+ postgres_app: tuple[Any, str], proxy: Proxy
375
+ ) -> None:
376
+ app, _dsn = postgres_app
377
+ await proxy.down()
378
+ async with app.router.lifespan_context(app), _client(app) as client:
379
+ assert (await client.get("/health")).status_code == 200 # alive: no crash loop
380
+ assert (await client.get("/ready")).status_code == 503
381
+ started = time.monotonic()
382
+ r = await client.post("/chat", json={"message": "hello"}, headers=AUTH)
383
+ assert r.status_code == 503 and "Reference: " in r.json()["detail"]
384
+ assert time.monotonic() - started < 1
385
+ assert (await client.get("/threads", headers=AUTH)).status_code == 503
386
+ await asyncio.sleep(1.5) # a few failed setup attempts
387
+ await proxy.up()
388
+ assert await _ready_within(client, 10) < 5
389
+ r = await client.post("/chat", json={"message": "hello"}, headers=AUTH)
390
+ assert parse_sse(r.text)[-1][0] == "message.end"
391
+
392
+
393
+ async def test_an_outage_fails_fast_logs_one_line_and_recovers_in_seconds(
394
+ postgres_app: tuple[Any, str], proxy: Proxy, caplog: pytest.LogCaptureFixture
395
+ ) -> None:
396
+ app, _dsn = postgres_app
397
+ async with app.router.lifespan_context(app), _client(app) as client:
398
+ await _ready_within(client, 10)
399
+ thread = str(uuid.uuid4())
400
+ r = await client.post("/chat", json={"message": "hello", "thread_id": thread}, headers=AUTH)
401
+ assert parse_sse(r.text)[-1][0] == "message.end"
402
+ await proxy.down() # the database restarts: every connection is cut
403
+ caplog.clear()
404
+ with caplog.at_level(logging.INFO):
405
+ durations = []
406
+ for _ in range(3):
407
+ started = time.monotonic()
408
+ r = await client.get("/threads", headers=AUTH)
409
+ durations.append(time.monotonic() - started)
410
+ assert r.status_code == 503, r.text
411
+ assert r.json()["detail"].startswith("Database unavailable. Reference: ")
412
+ assert (await client.get("/ready")).status_code == 503
413
+ # The first request finds out (bounded by the pool timeout), the next ones fail fast.
414
+ assert durations[0] < 7 and max(durations[1:]) < 3.5, durations
415
+ errors = [r for r in caplog.records if r.levelno >= logging.ERROR]
416
+ assert errors == [], [r.getMessage() for r in errors]
417
+ assert not any(r.exc_info for r in caplog.records if "unavailable" in r.getMessage())
418
+ await proxy.up()
419
+ assert await _ready_within(client, 15) < 8
420
+ r = await client.post("/chat", json={"message": "again", "thread_id": thread}, headers=AUTH)
421
+ assert parse_sse(r.text)[-1][0] == "message.end"
422
+
423
+
424
+ async def test_a_run_that_loses_its_lease_stops_before_writing(
425
+ postgres_app: tuple[Any, str], use_test_tools
426
+ ) -> None:
427
+ """Another replica took the thread over (the lease expired): the run must not write."""
428
+ from langchain_core.tools import tool
429
+
430
+ app, dsn = postgres_app
431
+ in_tool = asyncio.Event()
432
+ loop = asyncio.get_running_loop()
433
+
434
+ @tool
435
+ def slow_probe(query: str) -> str:
436
+ """Test-only tool: answers after 3 s."""
437
+ loop.call_soon_threadsafe(in_tool.set)
438
+ time.sleep(3)
439
+ return "late result"
440
+
441
+ async with app.router.lifespan_context(app), _client(app) as client:
442
+ await _ready_within(client, 10)
443
+ use_test_tools(slow_probe)
444
+ thread = str(uuid.uuid4())
445
+ run = asyncio.create_task(
446
+ client.post(
447
+ "/chat",
448
+ json={"message": "Run the slow probe for Paris", "thread_id": thread},
449
+ headers=AUTH,
450
+ )
451
+ )
452
+ await asyncio.wait_for(in_tool.wait(), 10)
453
+ # What a replica taking over an expired lease does to the row.
454
+ await _sql(
455
+ dsn,
456
+ "UPDATE thread_locks SET owner = 'other-replica', token = token + 1000 "
457
+ "WHERE thread_id = %s",
458
+ (thread,),
459
+ )
460
+ started = time.monotonic()
461
+ events = parse_sse((await run).text)
462
+ assert time.monotonic() - started < 2.5 # stopped, not waiting for the tool
463
+ assert [e for e, _ in events] == ["message.start", "tool.call", "error"]
464
+ assert events[-1][1]["code"] == "unavailable"
465
+ await asyncio.sleep(3.5) # the tool thread finishes: its result must not land
466
+ rows = await _sql(
467
+ dsn, "SELECT count(*) FROM checkpoint_writes WHERE thread_id = %s", (thread,)
468
+ )
469
+ writes_after = rows[0][0]
470
+ from {{cookiecutter.agent_directory}}.agent import graph
471
+
472
+ state = await graph.checkpointer.aget_tuple({"configurable": {"thread_id": thread}})
473
+ messages = state.checkpoint["channel_values"]["messages"]
474
+ assert [m.type for m in messages] == ["human", "ai"] # no tool result, no repair
475
+ rows = await _sql(dsn, "SELECT status FROM runs WHERE thread_id = %s", (thread,))
476
+ assert rows == [("interrupted",)]
477
+ # Nothing arrived later either.
478
+ await asyncio.sleep(0.5)
479
+ rows = await _sql(
480
+ dsn, "SELECT count(*) FROM checkpoint_writes WHERE thread_id = %s", (thread,)
481
+ )
482
+ assert rows[0][0] == writes_after
483
+
484
+
485
+ async def test_a_database_outage_mid_tool_call_ends_the_run_fast_and_the_thread_recovers(
486
+ postgres_app: tuple[Any, str], proxy: Proxy, use_test_tools
487
+ ) -> None:
488
+ from langchain_core.tools import tool
489
+
490
+ app, dsn = postgres_app
491
+ in_tool = asyncio.Event()
492
+ loop = asyncio.get_running_loop()
493
+
494
+ @tool
495
+ def slow_probe(query: str) -> str:
496
+ """Test-only tool: slow for Paris."""
497
+ if "Paris" in query:
498
+ loop.call_soon_threadsafe(in_tool.set)
499
+ time.sleep(2)
500
+ return "sunny"
501
+
502
+ async with app.router.lifespan_context(app), _client(app) as client:
503
+ await _ready_within(client, 10)
504
+ use_test_tools(slow_probe)
505
+ thread = str(uuid.uuid4())
506
+ run = asyncio.create_task(
507
+ client.post(
508
+ "/chat",
509
+ json={"message": "Run the slow probe for Paris", "thread_id": thread},
510
+ headers=AUTH,
511
+ )
512
+ )
513
+ await asyncio.wait_for(in_tool.wait(), 10)
514
+ await proxy.down()
515
+ started = time.monotonic()
516
+ events = parse_sse((await run).text)
517
+ assert time.monotonic() - started < 10 # not a minute of pool timeouts
518
+ assert events[-1][0] == "error" and events[-1][1]["code"] == "unavailable"
519
+ await proxy.up()
520
+ await _ready_within(client, 15)
521
+ deadline = time.monotonic() + 10
522
+ while True: # the lease of the stopped run is released in the background
523
+ r = await client.post(
524
+ "/chat", json={"message": "hello", "thread_id": thread}, headers=AUTH
525
+ )
526
+ if r.status_code != 409 or time.monotonic() > deadline:
527
+ break
528
+ await asyncio.sleep(0.3)
529
+ assert parse_sse(r.text)[-1][0] == "message.end", r.text
530
+ messages = (await client.get(f"/threads/{thread}/messages", headers=AUTH)).json()
531
+ assert [m["role"] for m in messages] == ["user", "assistant", "tool", "user", "assistant"]
532
+ assert (
533
+ messages[2]["is_error"]
534
+ and messages[2]["tool_call_id"] == (messages[1]["tool_calls"][0]["id"])
535
+ )
536
+ # Both runs are on record: the stopped one once the database was back.
537
+ deadline = time.monotonic() + 10
538
+ while True:
539
+ rows = await _sql(
540
+ dsn, "SELECT status FROM runs WHERE thread_id = %s ORDER BY created_at", (thread,)
541
+ )
542
+ if rows[:1] == [("interrupted",)] or time.monotonic() > deadline:
543
+ break
544
+ await asyncio.sleep(0.3)
545
+ assert rows == [("interrupted",), ("ok",)]
546
+
547
+
548
+ # --- a crash mid tool call ---------------------------------------------------------------------
549
+
550
+ # The server serves a graph with one test-only tool (never the project's own),
551
+ # slow when asked about "slow".
552
+ SERVER_SCRIPT = """
553
+ import os, sys, time
554
+ from langchain.agents import create_agent
555
+ from langchain_core.tools import tool
556
+ from {agent} import agent
557
+ from {agent}.app_utils import a2a, chat, run_locks
558
+ run_locks.LEASE_TTL_S, run_locks.RENEW_EVERY_S, run_locks.LOCAL_VALIDITY_S = 2.0, 0.2, 1.0
559
+ chat.RECONCILE_INTERVAL_S, chat.RECONCILE_GRACE_S = 0.5, 0.0
560
+ a2a.TASK_SWEEP_INTERVAL_S = 0.5
561
+ @tool
562
+ def probe(query: str) -> str:
563
+ \"\"\"Test-only tool: reports what it was asked about.\"\"\"
564
+ if "slow" in query:
565
+ time.sleep(60)
566
+ return "probe reading for " + query
567
+ agent.graph = create_agent(
568
+ model=agent.get_model(), tools=[probe], system_prompt=agent.SYSTEM_PROMPT,
569
+ middleware=agent.middleware(), context_schema=agent.AgentContext, name="test-agent",
570
+ ).with_config(dict(recursion_limit=agent.recursion_limit()))
571
+ import uvicorn
572
+ from {agent}.fast_api_app import app
573
+ uvicorn.run(app, host="127.0.0.1", port=int(sys.argv[1]), log_config=None)
574
+ """
575
+
576
+
577
+ def _free_port() -> int:
578
+ with socket.socket() as s:
579
+ s.bind(("127.0.0.1", 0))
580
+ return s.getsockname()[1]
581
+
582
+
583
+ def _start_server(dsn: str, port: int, log: Path) -> subprocess.Popen[bytes]:
584
+ env = {**os.environ, "CHECKPOINTER": "postgres", "POSTGRES_DSN": dsn, "LOG_FORMAT": "text"}
585
+ process = subprocess.Popen(
586
+ [sys.executable, "-c", SERVER_SCRIPT.format(agent=AGENT_DIR), str(port)],
587
+ cwd=PROJECT,
588
+ env=env,
589
+ stdout=log.open("ab"),
590
+ stderr=subprocess.STDOUT,
591
+ )
592
+ deadline = time.monotonic() + 30
593
+ while time.monotonic() < deadline:
594
+ with contextlib.suppress(httpx.HTTPError):
595
+ if httpx.get(f"http://127.0.0.1:{port}/ready", timeout=1).status_code == 200:
596
+ return process
597
+ time.sleep(0.2)
598
+ process.kill()
599
+ raise AssertionError("the server did not get ready: " + log.read_text()[-2000:])
600
+
601
+
602
+ def test_a_crash_mid_tool_call_leaves_a_usable_thread_and_an_interrupted_run(
603
+ dsn: str, tmp_path: Path
604
+ ) -> None:
605
+ port = _free_port()
606
+ base = f"http://127.0.0.1:{port}"
607
+ log = tmp_path / "server.log"
608
+ thread = f"crash-{uuid.uuid4().hex[:8]}"
609
+ server = _start_server(dsn, port, log)
610
+ try:
611
+ body = {"thread_id": thread, "message": "Run the probe for slow city"}
612
+ with contextlib.suppress(httpx.HTTPError):
613
+ with httpx.stream("POST", f"{base}/chat", json=body, headers=AUTH, timeout=30) as r:
614
+ for line in r.iter_lines():
615
+ if line.startswith("event: tool.call"):
616
+ time.sleep(1.0) # the model step's checkpoint lands
617
+ server.send_signal(signal.SIGKILL) # OOM kill, node loss, ...
618
+ server.wait()
619
+ break
620
+ server = _start_server(dsn, port, log)
621
+ crashed = httpx.get(f"{base}/threads/{thread}/messages", headers=AUTH).json()
622
+ assert [m["role"] for m in crashed] == ["user", "assistant"]
623
+ for message in ("hello", "hello again"):
624
+ deadline = time.monotonic() + 10
625
+ while True:
626
+ r = httpx.post(
627
+ f"{base}/chat", json={"thread_id": thread, "message": message}, headers=AUTH
628
+ )
629
+ # The dead process's lease holds the thread until it expires (2 s here).
630
+ if r.status_code != 409 or time.monotonic() > deadline:
631
+ break
632
+ time.sleep(0.3)
633
+ events = parse_sse(r.text)
634
+ assert events[-1][0] == "message.end", events
635
+ messages = httpx.get(f"{base}/threads/{thread}/messages", headers=AUTH).json()
636
+ roles = [m["role"] for m in messages]
637
+ # The open call got its result right after it, before the next turn.
638
+ assert roles == ["user", "assistant", "tool", "user", "assistant", "user", "assistant"]
639
+ assert messages[2]["is_error"] and "interrupted" in messages[2]["content"]
640
+ assert messages[2]["tool_call_id"] == messages[1]["tool_calls"][0]["id"]
641
+ # The killed run is on record, as interrupted, once its lease has expired.
642
+ deadline = time.monotonic() + 15
643
+ while True:
644
+ statuses = asyncio.run(
645
+ _sql(
646
+ dsn,
647
+ "SELECT status FROM runs WHERE thread_id = %s ORDER BY created_at",
648
+ (thread,),
649
+ )
650
+ )
651
+ if (statuses and statuses[0] == ("interrupted",)) or time.monotonic() > deadline:
652
+ break
653
+ time.sleep(0.5)
654
+ assert statuses == [("interrupted",), ("ok",), ("ok",)], statuses
655
+ metrics_text = httpx.get(f"{base}/metrics").text
656
+ assert 'agent_runs_total{status="interrupted"} 1.0' in metrics_text
657
+ finally:
658
+ server.terminate()
659
+ with contextlib.suppress(subprocess.TimeoutExpired):
660
+ server.wait(10)
661
+ if server.poll() is None:
662
+ server.kill()
663
+
664
+
665
+ # --- A2A tasks across replicas -----------------------------------------------------------------
666
+
667
+ A2A_PATH = f"/a2a/{AGENT_DIR}"
668
+
669
+
670
+ def _a2a(base: str, method: str, params: dict[str, Any]) -> dict[str, Any]:
671
+ """One JSON-RPC call on a new connection (a load balancer may send each to any replica)."""
672
+ r = httpx.post(
673
+ f"{base}{A2A_PATH}",
674
+ json={"jsonrpc": "2.0", "id": uuid.uuid4().hex, "method": method, "params": params},
675
+ headers={**AUTH, "A2A-Version": "1.0"},
676
+ timeout=30,
677
+ )
678
+ assert r.status_code == 200, r.text
679
+ return r.json()
680
+
681
+
682
+ def _say(text: str, **configuration: Any) -> dict[str, Any]:
683
+ message = {"messageId": uuid.uuid4().hex, "role": "ROLE_USER", "parts": [{"text": text}]}
684
+ return {"message": message, "configuration": configuration}
685
+
686
+
687
+ def _stop(server: subprocess.Popen[bytes]) -> None:
688
+ server.terminate()
689
+ with contextlib.suppress(subprocess.TimeoutExpired):
690
+ server.wait(10)
691
+ if server.poll() is None:
692
+ server.kill()
693
+
694
+
695
+ def test_a2a_tasks_are_seen_by_every_replica_and_survive_a_crash(dsn: str, tmp_path: Path) -> None:
696
+ ports = [_free_port(), _free_port()]
697
+ a, b = (f"http://127.0.0.1:{port}" for port in ports)
698
+ log = tmp_path / "server.log"
699
+ servers = [_start_server(dsn, port, log) for port in ports]
700
+ try:
701
+ task = _a2a(a, "SendMessage", _say("hello"))["result"]["task"]
702
+ assert task["status"]["state"] == "TASK_STATE_COMPLETED", task
703
+ # Another replica has the task, its reply and its listing (-32001 there before).
704
+ seen = _a2a(b, "GetTask", {"id": task["id"]})["result"]
705
+ assert seen["id"] == task["id"] and seen["artifacts"] == task["artifacts"]
706
+ listed = _a2a(b, "ListTasks", {"contextId": task["contextId"]})["result"]
707
+ assert [t["id"] for t in listed["tasks"]] == [task["id"]]
708
+ # The replica that ran it is killed; its replacement still has the task.
709
+ servers[0].send_signal(signal.SIGKILL)
710
+ servers[0].wait()
711
+ servers[0] = _start_server(dsn, ports[0], log)
712
+ again = _a2a(a, "GetTask", {"id": task["id"]})["result"]
713
+ assert again["status"]["state"] == "TASK_STATE_COMPLETED"
714
+ # Deleting the conversation on one replica deletes its tasks for every replica.
715
+ assert httpx.delete(f"{b}/threads/{task['contextId']}", headers=AUTH).status_code == 204
716
+ assert _a2a(a, "GetTask", {"id": task["id"]})["error"]["code"] == -32001
717
+ finally:
718
+ for server in servers:
719
+ _stop(server)
720
+
721
+
722
+ def test_an_a2a_task_whose_replica_dies_is_failed_and_not_canceled_elsewhere(
723
+ dsn: str, tmp_path: Path
724
+ ) -> None:
725
+ ports = [_free_port(), _free_port()]
726
+ a, b = (f"http://127.0.0.1:{port}" for port in ports)
727
+ log = tmp_path / "server.log"
728
+ servers = [_start_server(dsn, port, log) for port in ports]
729
+ try:
730
+ request = _say("Run the probe for slow city", returnImmediately=True)
731
+ task = _a2a(a, "SendMessage", request)["result"]["task"]
732
+ deadline = time.monotonic() + 10
733
+ while not asyncio.run(
734
+ _sql(
735
+ dsn,
736
+ "SELECT 1 FROM thread_locks WHERE thread_id = %s AND expires_at > now()",
737
+ (task["contextId"],),
738
+ )
739
+ ):
740
+ assert time.monotonic() < deadline, "the run never took its thread"
741
+ time.sleep(0.1)
742
+ # The run is in replica a: a cancel reaching b is refused, not reported as done.
743
+ refused = _a2a(b, "CancelTask", {"id": task["id"]})
744
+ assert refused.get("error", {}).get("code") == -32002, refused
745
+ assert "another replica" in refused["error"]["message"]
746
+ # So is a subscription there: the run's events happen in a only.
747
+ subscribe = {"jsonrpc": "2.0", "id": "s", "method": "SubscribeToTask"}
748
+ r = httpx.post(
749
+ f"{b}{A2A_PATH}",
750
+ json={**subscribe, "params": {"id": task["id"]}},
751
+ headers={**AUTH, "A2A-Version": "1.0"},
752
+ timeout=10,
753
+ )
754
+ assert '"code":-32004' in r.text.replace(" ", ""), r.text
755
+ assert _a2a(b, "GetTask", {"id": task["id"]})["result"]["status"]["state"] == (
756
+ "TASK_STATE_WORKING"
757
+ )
758
+ servers[0].send_signal(signal.SIGKILL) # the replica running it dies mid-run
759
+ servers[0].wait()
760
+ deadline = time.monotonic() + 20
761
+ while True:
762
+ status = _a2a(b, "GetTask", {"id": task["id"]})["result"]["status"]
763
+ if status["state"] != "TASK_STATE_WORKING" or time.monotonic() > deadline:
764
+ break
765
+ time.sleep(0.3)
766
+ assert status["state"] == "TASK_STATE_FAILED", status
767
+ assert "Send the message again" in status["message"]["parts"][0]["text"]
768
+ finally:
769
+ for server in servers:
770
+ _stop(server)