graph-agents-cli 0.3.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- graph_agents_cli/__init__.py +26 -0
- graph_agents_cli/_api_policy.py +2145 -0
- graph_agents_cli/_approvals.py +400 -0
- graph_agents_cli/_build.py +186 -0
- graph_agents_cli/_build_info.json +7 -0
- graph_agents_cli/_chat_client.py +462 -0
- graph_agents_cli/_click.py +157 -0
- graph_agents_cli/_defaults.py +139 -0
- graph_agents_cli/_experiments.py +64 -0
- graph_agents_cli/_http.py +192 -0
- graph_agents_cli/_output.py +83 -0
- graph_agents_cli/_project.py +462 -0
- graph_agents_cli/_remote.py +220 -0
- graph_agents_cli/_response_schema.py +264 -0
- graph_agents_cli/_runner.py +319 -0
- graph_agents_cli/_skills_check.py +274 -0
- graph_agents_cli/_tools.py +189 -0
- graph_agents_cli/_trust.py +66 -0
- graph_agents_cli/api/__init__.py +15 -0
- graph_agents_cli/api/_changes.py +506 -0
- graph_agents_cli/api/_files.py +658 -0
- graph_agents_cli/api/cmd_api.py +2480 -0
- graph_agents_cli/deploy/__init__.py +15 -0
- graph_agents_cli/deploy/_config.py +171 -0
- graph_agents_cli/deploy/_image.py +128 -0
- graph_agents_cli/deploy/_kube.py +286 -0
- graph_agents_cli/deploy/_modes.py +234 -0
- graph_agents_cli/deploy/_preflight.py +370 -0
- graph_agents_cli/deploy/_values.py +168 -0
- graph_agents_cli/deploy/cmd_deploy.py +1866 -0
- graph_agents_cli/deploy/gitops.py +562 -0
- graph_agents_cli/deploy/local_load.py +273 -0
- graph_agents_cli/dev/__init__.py +13 -0
- graph_agents_cli/dev/cmd_build.py +131 -0
- graph_agents_cli/dev/cmd_install.py +78 -0
- graph_agents_cli/dev/cmd_lint.py +119 -0
- graph_agents_cli/dev/cmd_playground.py +297 -0
- graph_agents_cli/dev/policy_check.py +1287 -0
- graph_agents_cli/eval/__init__.py +22 -0
- graph_agents_cli/eval/_client.py +670 -0
- graph_agents_cli/eval/_common.py +177 -0
- graph_agents_cli/eval/_judge.py +168 -0
- graph_agents_cli/eval/_judge_runner.py +238 -0
- graph_agents_cli/eval/_paths.py +212 -0
- graph_agents_cli/eval/checks.py +581 -0
- graph_agents_cli/eval/cmd_analyze.py +278 -0
- graph_agents_cli/eval/cmd_compare.py +284 -0
- graph_agents_cli/eval/cmd_eval_group.py +80 -0
- graph_agents_cli/eval/cmd_generate.py +558 -0
- graph_agents_cli/eval/cmd_grade.py +466 -0
- graph_agents_cli/eval/cmd_metric.py +156 -0
- graph_agents_cli/eval/cmd_run.py +370 -0
- graph_agents_cli/eval/cmd_submit.py +400 -0
- graph_agents_cli/eval/config.py +435 -0
- graph_agents_cli/eval/dataset.py +350 -0
- graph_agents_cli/eval/gate.py +420 -0
- graph_agents_cli/eval/transcript.py +192 -0
- graph_agents_cli/extension/__init__.py +13 -0
- graph_agents_cli/extension/_compat.py +86 -0
- graph_agents_cli/extension/_loader.py +293 -0
- graph_agents_cli/extension/_manifest.py +135 -0
- graph_agents_cli/extension/_overrides.py +195 -0
- graph_agents_cli/extension/_paths.py +91 -0
- graph_agents_cli/extension/_refs.py +193 -0
- graph_agents_cli/extension/_resolver.py +453 -0
- graph_agents_cli/extension/_schema.py +106 -0
- graph_agents_cli/extension/_spec.py +253 -0
- graph_agents_cli/extension/_sync.py +102 -0
- graph_agents_cli/extension/_trust.py +58 -0
- graph_agents_cli/extension/cmd_extension_add.py +259 -0
- graph_agents_cli/extension/cmd_extension_group.py +57 -0
- graph_agents_cli/extension/cmd_extension_list.py +56 -0
- graph_agents_cli/extension/cmd_extension_remove.py +61 -0
- graph_agents_cli/extension/cmd_extension_update.py +195 -0
- graph_agents_cli/info/__init__.py +13 -0
- graph_agents_cli/info/cmd_info.py +222 -0
- graph_agents_cli/infra/__init__.py +15 -0
- graph_agents_cli/infra/checks.py +1169 -0
- graph_agents_cli/infra/cmd_infra.py +103 -0
- graph_agents_cli/main.py +591 -0
- graph_agents_cli/peer/__init__.py +15 -0
- graph_agents_cli/peer/_generate.py +254 -0
- graph_agents_cli/peer/cmd_peer.py +1151 -0
- graph_agents_cli/run/__init__.py +13 -0
- graph_agents_cli/run/_local_server.py +1157 -0
- graph_agents_cli/run/_signals.py +141 -0
- graph_agents_cli/run/cmd_approvals.py +530 -0
- graph_agents_cli/run/cmd_run.py +1421 -0
- graph_agents_cli/scaffold/__init__.py +19 -0
- graph_agents_cli/scaffold/agents/README.md +24 -0
- graph_agents_cli/scaffold/agents/empty_py/.template/templateconfig.yaml +22 -0
- graph_agents_cli/scaffold/agents/langgraph/.env.example +292 -0
- graph_agents_cli/scaffold/agents/langgraph/.template/templateconfig.yaml +28 -0
- graph_agents_cli/scaffold/agents/langgraph/Dockerfile +59 -0
- graph_agents_cli/scaffold/agents/langgraph/Dockerfile.langgraph-server +59 -0
- graph_agents_cli/scaffold/agents/langgraph/README.md +571 -0
- graph_agents_cli/scaffold/agents/langgraph/api-policy.yaml +60 -0
- graph_agents_cli/scaffold/agents/langgraph/app/__init__.py +20 -0
- graph_agents_cli/scaffold/agents/langgraph/app/agent.py +174 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/__init__.py +15 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/a2a.py +2162 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/a2a_client.py +1167 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/api_client.py +4220 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/approvals.py +1349 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/auth.py +1986 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/chat.py +2962 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/checkpointer.py +432 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/content.py +569 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/db.py +580 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/limits.py +203 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/metrics.py +231 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/middleware.py +361 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/model.py +611 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/playground.py +230 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/run_locks.py +459 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/structured.py +755 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/telemetry.py +681 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/threads.py +493 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/token_exchange.py +959 -0
- graph_agents_cli/scaffold/agents/langgraph/app/fast_api_app.py +770 -0
- graph_agents_cli/scaffold/agents/langgraph/app/policies/__init__.py +55 -0
- graph_agents_cli/scaffold/agents/langgraph/app/policies/custom.py +97 -0
- graph_agents_cli/scaffold/agents/langgraph/app/tools/__init__.py +46 -0
- graph_agents_cli/scaffold/agents/langgraph/app/tools/example_api.py +92 -0
- graph_agents_cli/scaffold/agents/langgraph/app/tools/weather.py +33 -0
- graph_agents_cli/scaffold/agents/langgraph/langgraph.json +14 -0
- graph_agents_cli/scaffold/agents/langgraph/pyproject.toml +78 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/conftest.py +376 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/eval/datasets/basic-dataset.json +53 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/eval/eval_config.yaml +32 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/approval_graph.py +137 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/fake_issuer.py +216 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/fake_openai.py +357 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_a2a_outcomes.py +569 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_a2a_relay.py +479 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_api_surface.py +812 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_approvals.py +1367 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_approvals_server.py +794 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_cross_actor.py +497 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_cross_actor_server.py +247 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_history_repair.py +278 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_model_apis.py +242 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_postgres.py +637 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_resilience_postgres.py +770 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_runtime_guardrails.py +854 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_server_e2e.py +340 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_server_runtime.py +989 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_structured_answers.py +584 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_structured_server.py +222 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_token_exchange_issuer.py +650 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/load_test/.results/.placeholder +0 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/load_test/README.md +22 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/load_test/conftest.py +21 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/load_test/load_test.py +81 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_a2a_client.py +824 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_a2a_scoping.py +724 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_api_client.py +1214 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_api_client_hardening.py +716 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_api_policy_rpc.py +767 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_approval_ledger.py +1536 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_fake_model.py +115 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_jwt_policy.py +991 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_limits.py +310 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_logging.py +148 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_logging_hardening.py +271 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_policy.py +378 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_resilience.py +610 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_server_auth.py +702 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_structured.py +673 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_telemetry.py +404 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_thread_listing.py +255 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_threads.py +268 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_token_exchange.py +1320 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_untrusted_content.py +393 -0
- graph_agents_cli/scaffold/agents/langgraph/uv-fastapi.lock +2084 -0
- graph_agents_cli/scaffold/agents/langgraph/uv-langgraph-server.lock +2106 -0
- graph_agents_cli/scaffold/agents/langgraph/{{cookiecutter.agent_guidance_filename}} +129 -0
- graph_agents_cli/scaffold/base_templates/_shared/graph-agents-cli-manifest.yaml +36 -0
- graph_agents_cli/scaffold/base_templates/python/.dockerignore +32 -0
- graph_agents_cli/scaffold/base_templates/python/.github/CODEOWNERS +30 -0
- graph_agents_cli/scaffold/base_templates/python/.github/agent.env +7 -0
- graph_agents_cli/scaffold/base_templates/python/.github/workflows/pr_checks.yaml +214 -0
- graph_agents_cli/scaffold/base_templates/python/.gitignore +209 -0
- graph_agents_cli/scaffold/base_templates/python/tests/unit/test_dummy.py +23 -0
- graph_agents_cli/scaffold/base_templates/python/{{cookiecutter.agent_guidance_filename}} +35 -0
- graph_agents_cli/scaffold/cmd_scaffold_group.py +49 -0
- graph_agents_cli/scaffold/commands/__init__.py +13 -0
- graph_agents_cli/scaffold/commands/create.py +1424 -0
- graph_agents_cli/scaffold/commands/enhance.py +1652 -0
- graph_agents_cli/scaffold/commands/upgrade.py +570 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/.github/agent.env +12 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/.github/workflows/promote-to-prod.yaml +371 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/.github/workflows/staging.yaml +450 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/argocd/application-dev.yaml +43 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/argocd/application-prod.yaml +41 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/argocd/application-staging.yaml +43 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/.helmignore +14 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/Chart.yaml +21 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/examples/networkpolicy.yaml +103 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/NOTES.txt +48 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/_helpers.tpl +189 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/certificate.yaml +15 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/configmap.yaml +10 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/deployment.yaml +199 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/hpa.yaml +22 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/httproute.yaml +30 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/ingress.yaml +39 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/networkpolicy.yaml +48 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/pdb.yaml +13 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/postgresql-secret.yaml +37 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/service.yaml +15 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/serviceaccount.yaml +13 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/servicemonitor.yaml +42 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values-dev.yaml +22 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values-prod.yaml +45 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values-staging.yaml +29 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values.yaml +396 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/tests/integration/test_chart.py +269 -0
- graph_agents_cli/scaffold/deployment_targets/none/README.md +5 -0
- graph_agents_cli/scaffold/deployment_targets/none/python/README.md +6 -0
- graph_agents_cli/scaffold/utils/__init__.py +13 -0
- graph_agents_cli/scaffold/utils/backup.py +212 -0
- graph_agents_cli/scaffold/utils/build_record.py +257 -0
- graph_agents_cli/scaffold/utils/cli_options.py +184 -0
- graph_agents_cli/scaffold/utils/fs.py +83 -0
- graph_agents_cli/scaffold/utils/generate_locks.py +214 -0
- graph_agents_cli/scaffold/utils/generation_metadata.py +88 -0
- graph_agents_cli/scaffold/utils/keyedit.py +768 -0
- graph_agents_cli/scaffold/utils/keymerge.py +537 -0
- graph_agents_cli/scaffold/utils/language.py +138 -0
- graph_agents_cli/scaffold/utils/lock_utils.py +94 -0
- graph_agents_cli/scaffold/utils/logging.py +77 -0
- graph_agents_cli/scaffold/utils/manifest.py +292 -0
- graph_agents_cli/scaffold/utils/merge.py +970 -0
- graph_agents_cli/scaffold/utils/merge3.py +216 -0
- graph_agents_cli/scaffold/utils/openapi_seed.py +199 -0
- graph_agents_cli/scaffold/utils/remote_template.py +376 -0
- graph_agents_cli/scaffold/utils/template.py +1352 -0
- graph_agents_cli/scaffold/utils/upgrade.py +894 -0
- graph_agents_cli/scaffold/utils/version.py +438 -0
- graph_agents_cli/secrets/__init__.py +15 -0
- graph_agents_cli/secrets/_apply.py +954 -0
- graph_agents_cli/secrets/_required.py +188 -0
- graph_agents_cli/secrets/cmd_secrets.py +211 -0
- graph_agents_cli/setup/__init__.py +13 -0
- graph_agents_cli/setup/_antigravity.py +221 -0
- graph_agents_cli/setup/cmd_auth.py +1030 -0
- graph_agents_cli/setup/cmd_dev_token.py +513 -0
- graph_agents_cli/setup/cmd_setup.py +428 -0
- graph_agents_cli/setup/cmd_update.py +140 -0
- graph_agents_cli/skills/__init__.py +13 -0
- graph_agents_cli/skills/_bundle.py +65 -0
- graph_agents_cli/skills/data/README.md +19 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/SKILL.md +357 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/references/github-settings.md +113 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/references/gitops.md +137 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/references/kubernetes.md +315 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/references/secrets.md +160 -0
- graph_agents_cli/skills/data/graph-agents-cli-eval/SKILL.md +303 -0
- graph_agents_cli/skills/data/graph-agents-cli-eval/references/dataset_schema.md +282 -0
- graph_agents_cli/skills/data/graph-agents-cli-eval/references/metrics-guide.md +143 -0
- graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/SKILL.md +659 -0
- graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/references/langchain-models.md +124 -0
- graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/references/langgraph.md +235 -0
- graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/references/template-contract.md +477 -0
- graph_agents_cli/skills/data/graph-agents-cli-observability/SKILL.md +231 -0
- graph_agents_cli/skills/data/graph-agents-cli-observability/references/langsmith.md +46 -0
- graph_agents_cli/skills/data/graph-agents-cli-observability/references/otel.md +59 -0
- graph_agents_cli/skills/data/graph-agents-cli-scaffold/SKILL.md +414 -0
- graph_agents_cli/skills/data/graph-agents-cli-scaffold/references/flags.md +134 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/SKILL.md +478 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/brainstorming.md +118 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/commands.md +419 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/extension.md +156 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/internals.md +67 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/spec-template.md +56 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/terminology.md +119 -0
- graph_agents_cli/system/__init__.py +15 -0
- graph_agents_cli/system/_apply.py +519 -0
- graph_agents_cli/system/_checks.py +1023 -0
- graph_agents_cli/system/_deploy.py +215 -0
- graph_agents_cli/system/_model.py +363 -0
- graph_agents_cli/system/_system.py +664 -0
- graph_agents_cli/system/_views.py +208 -0
- graph_agents_cli/system/cmd_system.py +423 -0
- graph_agents_cli-0.3.1.dist-info/METADATA +162 -0
- graph_agents_cli-0.3.1.dist-info/RECORD +291 -0
- graph_agents_cli-0.3.1.dist-info/WHEEL +4 -0
- graph_agents_cli-0.3.1.dist-info/entry_points.txt +2 -0
- graph_agents_cli-0.3.1.dist-info/licenses/LICENSE +201 -0
- graph_agents_cli-0.3.1.dist-info/licenses/NOTICE +19 -0
|
@@ -0,0 +1,770 @@
|
|
|
1
|
+
# Copyright 2026 graph-agents-cli contributors
|
|
2
|
+
#
|
|
3
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
4
|
+
# you may not use this file except in compliance with the License.
|
|
5
|
+
# You may obtain a copy of the License at
|
|
6
|
+
#
|
|
7
|
+
# https://www.apache.org/licenses/LICENSE-2.0
|
|
8
|
+
#
|
|
9
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
10
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
11
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
12
|
+
# See the License for the specific language governing permissions and
|
|
13
|
+
# limitations under the License.
|
|
14
|
+
|
|
15
|
+
"""Failures against a real Postgres: lost replicas, dropped sessions, outages, crashes,
|
|
16
|
+
and A2A tasks across replicas.
|
|
17
|
+
|
|
18
|
+
Opt-in: set `TEST_POSTGRES_DSN` to the URL of a server where the user may
|
|
19
|
+
create databases (each test gets a fresh one, dropped afterwards). Outages are
|
|
20
|
+
made with a TCP proxy in front of that server, which the tests stop, restart
|
|
21
|
+
and cut; crashes with a real server process killed with SIGKILL.
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
from __future__ import annotations
|
|
25
|
+
|
|
26
|
+
import asyncio
|
|
27
|
+
import contextlib
|
|
28
|
+
import json
|
|
29
|
+
import logging
|
|
30
|
+
import os
|
|
31
|
+
import signal
|
|
32
|
+
import socket
|
|
33
|
+
import subprocess
|
|
34
|
+
import sys
|
|
35
|
+
import time
|
|
36
|
+
import uuid
|
|
37
|
+
from collections.abc import AsyncIterator
|
|
38
|
+
from pathlib import Path
|
|
39
|
+
from typing import Any
|
|
40
|
+
from urllib.parse import urlsplit
|
|
41
|
+
|
|
42
|
+
os.environ.update(
|
|
43
|
+
{
|
|
44
|
+
"MODEL_PROVIDER": "fake",
|
|
45
|
+
"MODEL_NAME": "fake",
|
|
46
|
+
"AUTH_POLICY": "shared-bearer",
|
|
47
|
+
"API_KEY": "test-key",
|
|
48
|
+
"APP_ENV": "dev",
|
|
49
|
+
"TRACING_ENABLED": "false",
|
|
50
|
+
"RUNTIME": "fastapi",
|
|
51
|
+
"APP_URL": "http://testserver",
|
|
52
|
+
}
|
|
53
|
+
)
|
|
54
|
+
|
|
55
|
+
import httpx
|
|
56
|
+
import pytest
|
|
57
|
+
|
|
58
|
+
from {{cookiecutter.agent_directory}}.app_utils import chat as chat_module
|
|
59
|
+
from {{cookiecutter.agent_directory}}.app_utils import run_locks
|
|
60
|
+
from {{cookiecutter.agent_directory}}.app_utils.checkpointer import POSTGRES
|
|
61
|
+
from {{cookiecutter.agent_directory}}.app_utils.db import Database, RunRecord, RunStore
|
|
62
|
+
from {{cookiecutter.agent_directory}}.app_utils.threads import LeaseLost, ThreadBusy, ThreadLocks
|
|
63
|
+
|
|
64
|
+
ADMIN_DSN = os.environ.get("TEST_POSTGRES_DSN", "")
|
|
65
|
+
pytestmark = pytest.mark.skipif(not ADMIN_DSN, reason="TEST_POSTGRES_DSN is not set")
|
|
66
|
+
AUTH = {"Authorization": "Bearer test-key"}
|
|
67
|
+
PROJECT = Path(__file__).resolve().parents[2]
|
|
68
|
+
AGENT_DIR = "{{cookiecutter.agent_directory}}"
|
|
69
|
+
# Short leases, so a lost replica is noticed in seconds.
|
|
70
|
+
FAST_LEASES = {"ttl_s": 2.0, "renew_every_s": 0.2, "validity_s": 1.0}
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _with_database(dsn: str, name: str) -> str:
|
|
74
|
+
return urlsplit(dsn)._replace(path=f"/{name}").geturl()
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _with_port(dsn: str, port: int) -> str:
|
|
78
|
+
parts = urlsplit(dsn)
|
|
79
|
+
userinfo = parts.netloc.rsplit("@", 1)[0] + "@" if "@" in parts.netloc else ""
|
|
80
|
+
return parts._replace(netloc=f"{userinfo}127.0.0.1:{port}").geturl()
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
@pytest.fixture
|
|
84
|
+
async def dsn() -> AsyncIterator[str]:
|
|
85
|
+
import psycopg
|
|
86
|
+
|
|
87
|
+
name = f"gac_test_{uuid.uuid4().hex[:12]}"
|
|
88
|
+
async with await psycopg.AsyncConnection.connect(ADMIN_DSN, autocommit=True) as admin:
|
|
89
|
+
await admin.execute(f'CREATE DATABASE "{name}"')
|
|
90
|
+
yield _with_database(ADMIN_DSN, name)
|
|
91
|
+
async with await psycopg.AsyncConnection.connect(ADMIN_DSN, autocommit=True) as admin:
|
|
92
|
+
await admin.execute(f'DROP DATABASE IF EXISTS "{name}" WITH (FORCE)')
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
async def _schema(dsn: str) -> None:
|
|
96
|
+
db = Database(POSTGRES, dsn)
|
|
97
|
+
await db.open()
|
|
98
|
+
await db.close()
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
async def _sql(dsn: str, sql: str, params: tuple[Any, ...] = ()) -> list[tuple[Any, ...]]:
|
|
102
|
+
import psycopg
|
|
103
|
+
|
|
104
|
+
async with await psycopg.AsyncConnection.connect(dsn, autocommit=True) as conn:
|
|
105
|
+
cur = await conn.execute(sql, params)
|
|
106
|
+
return list(await cur.fetchall()) if cur.description else []
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
async def _acquire_within(locks: ThreadLocks, thread_id: str, seconds: float) -> float:
|
|
110
|
+
"""Seconds until `locks` gets the thread (fails after `seconds`)."""
|
|
111
|
+
start = time.monotonic()
|
|
112
|
+
while True:
|
|
113
|
+
try:
|
|
114
|
+
lease = await locks.acquire(thread_id)
|
|
115
|
+
except ThreadBusy:
|
|
116
|
+
if time.monotonic() - start > seconds:
|
|
117
|
+
raise
|
|
118
|
+
await asyncio.sleep(0.1)
|
|
119
|
+
continue
|
|
120
|
+
await lease.release()
|
|
121
|
+
return time.monotonic() - start
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
# --- run leases ----------------------------------------------------------------------------
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
async def test_a_replica_lost_without_closing_its_connection_frees_its_threads(
|
|
128
|
+
dsn: str,
|
|
129
|
+
) -> None:
|
|
130
|
+
"""A frozen or partitioned replica: its session stays open, it just stops renewing."""
|
|
131
|
+
await _schema(dsn)
|
|
132
|
+
lost_replica, other = ThreadLocks(dsn, **FAST_LEASES), ThreadLocks(dsn, **FAST_LEASES)
|
|
133
|
+
try:
|
|
134
|
+
lease = await lost_replica.acquire("t1")
|
|
135
|
+
stopped: list[str] = []
|
|
136
|
+
lease.on_lost(lambda: stopped.append("stopped"))
|
|
137
|
+
with pytest.raises(ThreadBusy):
|
|
138
|
+
await other.acquire("t1")
|
|
139
|
+
heartbeat = lost_replica._task
|
|
140
|
+
assert heartbeat is not None
|
|
141
|
+
heartbeat.cancel() # frozen: no renewals, the connection stays open
|
|
142
|
+
waited = await _acquire_within(other, "t1", 10)
|
|
143
|
+
assert 1.0 < waited < 6.0 # the 2 s lease, not the OS's 2 h TCP keepalive
|
|
144
|
+
# The lost replica may no longer write the thread, and knows it.
|
|
145
|
+
with pytest.raises(LeaseLost):
|
|
146
|
+
lost_replica.fence("t1")
|
|
147
|
+
assert stopped == ["stopped"] and lease.lost
|
|
148
|
+
# Back from the partition: its release does not free what it no longer holds.
|
|
149
|
+
taken = await other.acquire("t1")
|
|
150
|
+
await lease.release()
|
|
151
|
+
third = ThreadLocks(dsn, **FAST_LEASES)
|
|
152
|
+
try:
|
|
153
|
+
with pytest.raises(ThreadBusy):
|
|
154
|
+
await third.acquire("t1")
|
|
155
|
+
finally:
|
|
156
|
+
await third.close()
|
|
157
|
+
await taken.release()
|
|
158
|
+
finally:
|
|
159
|
+
await lost_replica.close()
|
|
160
|
+
await other.close()
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
async def test_replicas_racing_for_a_thread_get_it_once(dsn: str) -> None:
|
|
164
|
+
await _schema(dsn)
|
|
165
|
+
replicas = [ThreadLocks(dsn, **FAST_LEASES) for _ in range(8)]
|
|
166
|
+
try:
|
|
167
|
+
results = await asyncio.gather(*(r.acquire("t1") for r in replicas), return_exceptions=True)
|
|
168
|
+
won = [r for r in results if not isinstance(r, BaseException)]
|
|
169
|
+
assert len(won) == 1
|
|
170
|
+
assert all(isinstance(r, ThreadBusy) for r in results if r not in won)
|
|
171
|
+
tokens = await _sql(dsn, "SELECT token FROM thread_locks WHERE thread_id = 't1'")
|
|
172
|
+
assert tokens == [(won[0].token,)]
|
|
173
|
+
await won[0].release()
|
|
174
|
+
# The next holder gets a new fencing token.
|
|
175
|
+
again = await replicas[3].acquire("t1")
|
|
176
|
+
assert again.token is not None and again.token > won[0].token
|
|
177
|
+
await again.release()
|
|
178
|
+
finally:
|
|
179
|
+
for r in replicas:
|
|
180
|
+
await r.close()
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
async def test_a_dropped_session_keeps_the_lease_and_its_run(dsn: str) -> None:
|
|
184
|
+
"""What a Postgres restart or failover, an admin kill or a proxy reset does to a session."""
|
|
185
|
+
await _schema(dsn)
|
|
186
|
+
holder, other = ThreadLocks(dsn, **FAST_LEASES), ThreadLocks(dsn, **FAST_LEASES)
|
|
187
|
+
try:
|
|
188
|
+
lease = await holder.acquire("t1")
|
|
189
|
+
rows = await _sql(
|
|
190
|
+
dsn,
|
|
191
|
+
"SELECT count(pg_terminate_backend(pid)) FROM pg_stat_activity "
|
|
192
|
+
"WHERE datname = current_database() AND pid <> pg_backend_pid()",
|
|
193
|
+
)
|
|
194
|
+
assert rows[0][0] >= 1
|
|
195
|
+
with pytest.raises(ThreadBusy): # the old advisory lock was gone at this point
|
|
196
|
+
await other.acquire("t1")
|
|
197
|
+
await asyncio.sleep(FAST_LEASES["validity_s"] * 2) # renewals reconnect
|
|
198
|
+
holder.fence("t1")
|
|
199
|
+
assert not lease.lost
|
|
200
|
+
with pytest.raises(ThreadBusy):
|
|
201
|
+
await other.acquire("t1")
|
|
202
|
+
await lease.release()
|
|
203
|
+
assert await _acquire_within(other, "t1", 2) < 1.0
|
|
204
|
+
finally:
|
|
205
|
+
await holder.close()
|
|
206
|
+
await other.close()
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
async def test_a_lease_whose_release_failed_is_released_in_the_background(
|
|
210
|
+
dsn: str, monkeypatch: pytest.MonkeyPatch
|
|
211
|
+
) -> None:
|
|
212
|
+
await _schema(dsn)
|
|
213
|
+
holder, other = ThreadLocks(dsn, **FAST_LEASES), ThreadLocks(dsn, **FAST_LEASES)
|
|
214
|
+
try:
|
|
215
|
+
execute = holder._execute
|
|
216
|
+
fail_release = {"t1", "t2"} # the first release of each fails
|
|
217
|
+
|
|
218
|
+
async def flaky(sql: str, params: dict[str, Any], **kwargs: Any) -> Any:
|
|
219
|
+
if sql.startswith("DELETE") and params.get("thread") in fail_release:
|
|
220
|
+
fail_release.discard(params["thread"])
|
|
221
|
+
raise OSError("connection refused")
|
|
222
|
+
return await execute(sql, params, **kwargs)
|
|
223
|
+
|
|
224
|
+
monkeypatch.setattr(holder, "_execute", flaky)
|
|
225
|
+
lease = await holder.acquire("t1")
|
|
226
|
+
await lease.release()
|
|
227
|
+
assert "t1" in holder._unreleased and "t1" not in holder.held
|
|
228
|
+
# Every replica gets it once the background retry went through...
|
|
229
|
+
assert await _acquire_within(other, "t1", 5) < 4
|
|
230
|
+
# ...and this process takes a leftover of its own back at once.
|
|
231
|
+
lease = await holder.acquire("t2")
|
|
232
|
+
await lease.release()
|
|
233
|
+
assert "t2" in holder._unreleased
|
|
234
|
+
again = await holder.acquire("t2")
|
|
235
|
+
assert "t2" not in holder._unreleased
|
|
236
|
+
await again.release()
|
|
237
|
+
assert await _acquire_within(other, "t2", 2) < 1
|
|
238
|
+
finally:
|
|
239
|
+
await holder.close()
|
|
240
|
+
await other.close()
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
async def test_runs_of_dead_processes_are_reconciled_as_interrupted(dsn: str) -> None:
|
|
244
|
+
await _schema(dsn)
|
|
245
|
+
db = Database(POSTGRES, dsn)
|
|
246
|
+
await db.open()
|
|
247
|
+
locks = ThreadLocks(dsn, **FAST_LEASES)
|
|
248
|
+
try:
|
|
249
|
+
runs = RunStore(db)
|
|
250
|
+
old = "2000-01-01T00:00:00+00:00"
|
|
251
|
+
for run_id, thread_id in (("dead", "t-dead"), ("alive", "t-alive")):
|
|
252
|
+
await runs.start(
|
|
253
|
+
RunRecord(
|
|
254
|
+
run_id=run_id,
|
|
255
|
+
thread_id=thread_id,
|
|
256
|
+
principal_hash="h",
|
|
257
|
+
model="m",
|
|
258
|
+
status="ok",
|
|
259
|
+
created_at=old,
|
|
260
|
+
)
|
|
261
|
+
)
|
|
262
|
+
lease = await locks.acquire("t-alive") # a run still holds its thread
|
|
263
|
+
replicas = await asyncio.gather(runs.reconcile(0), runs.reconcile(0))
|
|
264
|
+
assert sorted(replicas, key=len) == [[], ["dead"]] # closed once across replicas
|
|
265
|
+
dead, alive = await runs.get("dead"), await runs.get("alive")
|
|
266
|
+
assert dead is not None and dead.status == "interrupted"
|
|
267
|
+
assert dead.error_type == "ProcessLost"
|
|
268
|
+
assert alive is not None and alive.status == "running"
|
|
269
|
+
await lease.release()
|
|
270
|
+
assert await runs.reconcile(0) == ["alive"]
|
|
271
|
+
finally:
|
|
272
|
+
await locks.close()
|
|
273
|
+
await db.close()
|
|
274
|
+
|
|
275
|
+
|
|
276
|
+
# --- a TCP proxy that can take the database away ---------------------------------------------
|
|
277
|
+
|
|
278
|
+
|
|
279
|
+
class Proxy:
|
|
280
|
+
"""Forwards 127.0.0.1:<port> to the test server; `down()` refuses and cuts every connection."""
|
|
281
|
+
|
|
282
|
+
def __init__(self, upstream: str) -> None:
|
|
283
|
+
parts = urlsplit(upstream)
|
|
284
|
+
self.host, self.upstream_port = parts.hostname or "127.0.0.1", parts.port or 5432
|
|
285
|
+
self.port = 0
|
|
286
|
+
self.server: asyncio.base_events.Server | None = None
|
|
287
|
+
self.writers: set[asyncio.StreamWriter] = set()
|
|
288
|
+
|
|
289
|
+
async def up(self) -> None:
|
|
290
|
+
self.server = await asyncio.start_server(self._handle, "127.0.0.1", self.port)
|
|
291
|
+
self.port = self.server.sockets[0].getsockname()[1]
|
|
292
|
+
|
|
293
|
+
async def down(self) -> None:
|
|
294
|
+
server, self.server = self.server, None
|
|
295
|
+
if server is not None:
|
|
296
|
+
server.close() # refuse new connections
|
|
297
|
+
for writer in list(self.writers): # cut the open ones
|
|
298
|
+
writer.close()
|
|
299
|
+
self.writers.clear()
|
|
300
|
+
if server is not None:
|
|
301
|
+
with contextlib.suppress(TimeoutError):
|
|
302
|
+
await asyncio.wait_for(server.wait_closed(), 5)
|
|
303
|
+
|
|
304
|
+
async def _handle(self, reader: asyncio.StreamReader, writer: asyncio.StreamWriter) -> None:
|
|
305
|
+
try:
|
|
306
|
+
up_reader, up_writer = await asyncio.open_connection(self.host, self.upstream_port)
|
|
307
|
+
except OSError:
|
|
308
|
+
writer.close()
|
|
309
|
+
return
|
|
310
|
+
self.writers.update((writer, up_writer))
|
|
311
|
+
|
|
312
|
+
async def pipe(src: asyncio.StreamReader, dst: asyncio.StreamWriter) -> None:
|
|
313
|
+
with contextlib.suppress(Exception):
|
|
314
|
+
while data := await src.read(65536):
|
|
315
|
+
dst.write(data)
|
|
316
|
+
await dst.drain()
|
|
317
|
+
dst.close()
|
|
318
|
+
|
|
319
|
+
await asyncio.gather(pipe(reader, up_writer), pipe(up_reader, writer))
|
|
320
|
+
self.writers.difference_update((writer, up_writer))
|
|
321
|
+
|
|
322
|
+
|
|
323
|
+
@pytest.fixture
|
|
324
|
+
async def proxy(dsn: str) -> AsyncIterator[Proxy]:
|
|
325
|
+
p = Proxy(dsn)
|
|
326
|
+
await p.up()
|
|
327
|
+
yield p
|
|
328
|
+
await p.down()
|
|
329
|
+
|
|
330
|
+
|
|
331
|
+
@pytest.fixture
|
|
332
|
+
def postgres_app(dsn: str, proxy: Proxy, monkeypatch: pytest.MonkeyPatch) -> tuple[Any, str]:
|
|
333
|
+
"""The app configured for Postgres through the proxy (enter its lifespan in the test)."""
|
|
334
|
+
from {{cookiecutter.agent_directory}}.fast_api_app import app
|
|
335
|
+
|
|
336
|
+
for name in ("RETENTION_DAYS", "RECURSION_LIMIT", "RUN_TIMEOUT_S", "SSE_HEARTBEAT_S"):
|
|
337
|
+
monkeypatch.delenv(name, raising=False)
|
|
338
|
+
monkeypatch.setenv("CHECKPOINTER", "postgres")
|
|
339
|
+
monkeypatch.setenv("POSTGRES_DSN", _with_port(dsn, proxy.port))
|
|
340
|
+
monkeypatch.setattr(run_locks, "LEASE_TTL_S", FAST_LEASES["ttl_s"])
|
|
341
|
+
monkeypatch.setattr(run_locks, "RENEW_EVERY_S", FAST_LEASES["renew_every_s"])
|
|
342
|
+
monkeypatch.setattr(run_locks, "LOCAL_VALIDITY_S", FAST_LEASES["validity_s"])
|
|
343
|
+
monkeypatch.setattr(chat_module, "INIT_RETRY_MAX_S", 0.5)
|
|
344
|
+
monkeypatch.setattr(chat_module, "RECONCILE_INTERVAL_S", 0.5)
|
|
345
|
+
return app, dsn
|
|
346
|
+
|
|
347
|
+
|
|
348
|
+
def _client(app: Any) -> httpx.AsyncClient:
|
|
349
|
+
transport = httpx.ASGITransport(app=app, raise_app_exceptions=False)
|
|
350
|
+
return httpx.AsyncClient(transport=transport, base_url="http://testserver", timeout=30)
|
|
351
|
+
|
|
352
|
+
|
|
353
|
+
def parse_sse(text: str) -> list[tuple[str, dict[str, Any]]]:
|
|
354
|
+
events: list[tuple[str, dict[str, Any]]] = []
|
|
355
|
+
event = None
|
|
356
|
+
for line in text.splitlines():
|
|
357
|
+
if line.startswith("event:"):
|
|
358
|
+
event = line[6:].strip()
|
|
359
|
+
elif line.startswith("data:") and event:
|
|
360
|
+
events.append((event, json.loads(line[5:].strip())))
|
|
361
|
+
event = None
|
|
362
|
+
return events
|
|
363
|
+
|
|
364
|
+
|
|
365
|
+
async def _ready_within(client: httpx.AsyncClient, seconds: float) -> float:
|
|
366
|
+
start = time.monotonic()
|
|
367
|
+
while (await client.get("/ready")).status_code != 200:
|
|
368
|
+
assert time.monotonic() - start < seconds, "not ready in time"
|
|
369
|
+
await asyncio.sleep(0.2)
|
|
370
|
+
return time.monotonic() - start
|
|
371
|
+
|
|
372
|
+
|
|
373
|
+
async def test_the_app_starts_while_the_database_is_down_and_gets_ready_after_it(
|
|
374
|
+
postgres_app: tuple[Any, str], proxy: Proxy
|
|
375
|
+
) -> None:
|
|
376
|
+
app, _dsn = postgres_app
|
|
377
|
+
await proxy.down()
|
|
378
|
+
async with app.router.lifespan_context(app), _client(app) as client:
|
|
379
|
+
assert (await client.get("/health")).status_code == 200 # alive: no crash loop
|
|
380
|
+
assert (await client.get("/ready")).status_code == 503
|
|
381
|
+
started = time.monotonic()
|
|
382
|
+
r = await client.post("/chat", json={"message": "hello"}, headers=AUTH)
|
|
383
|
+
assert r.status_code == 503 and "Reference: " in r.json()["detail"]
|
|
384
|
+
assert time.monotonic() - started < 1
|
|
385
|
+
assert (await client.get("/threads", headers=AUTH)).status_code == 503
|
|
386
|
+
await asyncio.sleep(1.5) # a few failed setup attempts
|
|
387
|
+
await proxy.up()
|
|
388
|
+
assert await _ready_within(client, 10) < 5
|
|
389
|
+
r = await client.post("/chat", json={"message": "hello"}, headers=AUTH)
|
|
390
|
+
assert parse_sse(r.text)[-1][0] == "message.end"
|
|
391
|
+
|
|
392
|
+
|
|
393
|
+
async def test_an_outage_fails_fast_logs_one_line_and_recovers_in_seconds(
|
|
394
|
+
postgres_app: tuple[Any, str], proxy: Proxy, caplog: pytest.LogCaptureFixture
|
|
395
|
+
) -> None:
|
|
396
|
+
app, _dsn = postgres_app
|
|
397
|
+
async with app.router.lifespan_context(app), _client(app) as client:
|
|
398
|
+
await _ready_within(client, 10)
|
|
399
|
+
thread = str(uuid.uuid4())
|
|
400
|
+
r = await client.post("/chat", json={"message": "hello", "thread_id": thread}, headers=AUTH)
|
|
401
|
+
assert parse_sse(r.text)[-1][0] == "message.end"
|
|
402
|
+
await proxy.down() # the database restarts: every connection is cut
|
|
403
|
+
caplog.clear()
|
|
404
|
+
with caplog.at_level(logging.INFO):
|
|
405
|
+
durations = []
|
|
406
|
+
for _ in range(3):
|
|
407
|
+
started = time.monotonic()
|
|
408
|
+
r = await client.get("/threads", headers=AUTH)
|
|
409
|
+
durations.append(time.monotonic() - started)
|
|
410
|
+
assert r.status_code == 503, r.text
|
|
411
|
+
assert r.json()["detail"].startswith("Database unavailable. Reference: ")
|
|
412
|
+
assert (await client.get("/ready")).status_code == 503
|
|
413
|
+
# The first request finds out (bounded by the pool timeout), the next ones fail fast.
|
|
414
|
+
assert durations[0] < 7 and max(durations[1:]) < 3.5, durations
|
|
415
|
+
errors = [r for r in caplog.records if r.levelno >= logging.ERROR]
|
|
416
|
+
assert errors == [], [r.getMessage() for r in errors]
|
|
417
|
+
assert not any(r.exc_info for r in caplog.records if "unavailable" in r.getMessage())
|
|
418
|
+
await proxy.up()
|
|
419
|
+
assert await _ready_within(client, 15) < 8
|
|
420
|
+
r = await client.post("/chat", json={"message": "again", "thread_id": thread}, headers=AUTH)
|
|
421
|
+
assert parse_sse(r.text)[-1][0] == "message.end"
|
|
422
|
+
|
|
423
|
+
|
|
424
|
+
async def test_a_run_that_loses_its_lease_stops_before_writing(
|
|
425
|
+
postgres_app: tuple[Any, str], use_test_tools
|
|
426
|
+
) -> None:
|
|
427
|
+
"""Another replica took the thread over (the lease expired): the run must not write."""
|
|
428
|
+
from langchain_core.tools import tool
|
|
429
|
+
|
|
430
|
+
app, dsn = postgres_app
|
|
431
|
+
in_tool = asyncio.Event()
|
|
432
|
+
loop = asyncio.get_running_loop()
|
|
433
|
+
|
|
434
|
+
@tool
|
|
435
|
+
def slow_probe(query: str) -> str:
|
|
436
|
+
"""Test-only tool: answers after 3 s."""
|
|
437
|
+
loop.call_soon_threadsafe(in_tool.set)
|
|
438
|
+
time.sleep(3)
|
|
439
|
+
return "late result"
|
|
440
|
+
|
|
441
|
+
async with app.router.lifespan_context(app), _client(app) as client:
|
|
442
|
+
await _ready_within(client, 10)
|
|
443
|
+
use_test_tools(slow_probe)
|
|
444
|
+
thread = str(uuid.uuid4())
|
|
445
|
+
run = asyncio.create_task(
|
|
446
|
+
client.post(
|
|
447
|
+
"/chat",
|
|
448
|
+
json={"message": "Run the slow probe for Paris", "thread_id": thread},
|
|
449
|
+
headers=AUTH,
|
|
450
|
+
)
|
|
451
|
+
)
|
|
452
|
+
await asyncio.wait_for(in_tool.wait(), 10)
|
|
453
|
+
# What a replica taking over an expired lease does to the row.
|
|
454
|
+
await _sql(
|
|
455
|
+
dsn,
|
|
456
|
+
"UPDATE thread_locks SET owner = 'other-replica', token = token + 1000 "
|
|
457
|
+
"WHERE thread_id = %s",
|
|
458
|
+
(thread,),
|
|
459
|
+
)
|
|
460
|
+
started = time.monotonic()
|
|
461
|
+
events = parse_sse((await run).text)
|
|
462
|
+
assert time.monotonic() - started < 2.5 # stopped, not waiting for the tool
|
|
463
|
+
assert [e for e, _ in events] == ["message.start", "tool.call", "error"]
|
|
464
|
+
assert events[-1][1]["code"] == "unavailable"
|
|
465
|
+
await asyncio.sleep(3.5) # the tool thread finishes: its result must not land
|
|
466
|
+
rows = await _sql(
|
|
467
|
+
dsn, "SELECT count(*) FROM checkpoint_writes WHERE thread_id = %s", (thread,)
|
|
468
|
+
)
|
|
469
|
+
writes_after = rows[0][0]
|
|
470
|
+
from {{cookiecutter.agent_directory}}.agent import graph
|
|
471
|
+
|
|
472
|
+
state = await graph.checkpointer.aget_tuple({"configurable": {"thread_id": thread}})
|
|
473
|
+
messages = state.checkpoint["channel_values"]["messages"]
|
|
474
|
+
assert [m.type for m in messages] == ["human", "ai"] # no tool result, no repair
|
|
475
|
+
rows = await _sql(dsn, "SELECT status FROM runs WHERE thread_id = %s", (thread,))
|
|
476
|
+
assert rows == [("interrupted",)]
|
|
477
|
+
# Nothing arrived later either.
|
|
478
|
+
await asyncio.sleep(0.5)
|
|
479
|
+
rows = await _sql(
|
|
480
|
+
dsn, "SELECT count(*) FROM checkpoint_writes WHERE thread_id = %s", (thread,)
|
|
481
|
+
)
|
|
482
|
+
assert rows[0][0] == writes_after
|
|
483
|
+
|
|
484
|
+
|
|
485
|
+
async def test_a_database_outage_mid_tool_call_ends_the_run_fast_and_the_thread_recovers(
|
|
486
|
+
postgres_app: tuple[Any, str], proxy: Proxy, use_test_tools
|
|
487
|
+
) -> None:
|
|
488
|
+
from langchain_core.tools import tool
|
|
489
|
+
|
|
490
|
+
app, dsn = postgres_app
|
|
491
|
+
in_tool = asyncio.Event()
|
|
492
|
+
loop = asyncio.get_running_loop()
|
|
493
|
+
|
|
494
|
+
@tool
|
|
495
|
+
def slow_probe(query: str) -> str:
|
|
496
|
+
"""Test-only tool: slow for Paris."""
|
|
497
|
+
if "Paris" in query:
|
|
498
|
+
loop.call_soon_threadsafe(in_tool.set)
|
|
499
|
+
time.sleep(2)
|
|
500
|
+
return "sunny"
|
|
501
|
+
|
|
502
|
+
async with app.router.lifespan_context(app), _client(app) as client:
|
|
503
|
+
await _ready_within(client, 10)
|
|
504
|
+
use_test_tools(slow_probe)
|
|
505
|
+
thread = str(uuid.uuid4())
|
|
506
|
+
run = asyncio.create_task(
|
|
507
|
+
client.post(
|
|
508
|
+
"/chat",
|
|
509
|
+
json={"message": "Run the slow probe for Paris", "thread_id": thread},
|
|
510
|
+
headers=AUTH,
|
|
511
|
+
)
|
|
512
|
+
)
|
|
513
|
+
await asyncio.wait_for(in_tool.wait(), 10)
|
|
514
|
+
await proxy.down()
|
|
515
|
+
started = time.monotonic()
|
|
516
|
+
events = parse_sse((await run).text)
|
|
517
|
+
assert time.monotonic() - started < 10 # not a minute of pool timeouts
|
|
518
|
+
assert events[-1][0] == "error" and events[-1][1]["code"] == "unavailable"
|
|
519
|
+
await proxy.up()
|
|
520
|
+
await _ready_within(client, 15)
|
|
521
|
+
deadline = time.monotonic() + 10
|
|
522
|
+
while True: # the lease of the stopped run is released in the background
|
|
523
|
+
r = await client.post(
|
|
524
|
+
"/chat", json={"message": "hello", "thread_id": thread}, headers=AUTH
|
|
525
|
+
)
|
|
526
|
+
if r.status_code != 409 or time.monotonic() > deadline:
|
|
527
|
+
break
|
|
528
|
+
await asyncio.sleep(0.3)
|
|
529
|
+
assert parse_sse(r.text)[-1][0] == "message.end", r.text
|
|
530
|
+
messages = (await client.get(f"/threads/{thread}/messages", headers=AUTH)).json()
|
|
531
|
+
assert [m["role"] for m in messages] == ["user", "assistant", "tool", "user", "assistant"]
|
|
532
|
+
assert (
|
|
533
|
+
messages[2]["is_error"]
|
|
534
|
+
and messages[2]["tool_call_id"] == (messages[1]["tool_calls"][0]["id"])
|
|
535
|
+
)
|
|
536
|
+
# Both runs are on record: the stopped one once the database was back.
|
|
537
|
+
deadline = time.monotonic() + 10
|
|
538
|
+
while True:
|
|
539
|
+
rows = await _sql(
|
|
540
|
+
dsn, "SELECT status FROM runs WHERE thread_id = %s ORDER BY created_at", (thread,)
|
|
541
|
+
)
|
|
542
|
+
if rows[:1] == [("interrupted",)] or time.monotonic() > deadline:
|
|
543
|
+
break
|
|
544
|
+
await asyncio.sleep(0.3)
|
|
545
|
+
assert rows == [("interrupted",), ("ok",)]
|
|
546
|
+
|
|
547
|
+
|
|
548
|
+
# --- a crash mid tool call ---------------------------------------------------------------------
|
|
549
|
+
|
|
550
|
+
# The server serves a graph with one test-only tool (never the project's own),
|
|
551
|
+
# slow when asked about "slow".
|
|
552
|
+
SERVER_SCRIPT = """
|
|
553
|
+
import os, sys, time
|
|
554
|
+
from langchain.agents import create_agent
|
|
555
|
+
from langchain_core.tools import tool
|
|
556
|
+
from {agent} import agent
|
|
557
|
+
from {agent}.app_utils import a2a, chat, run_locks
|
|
558
|
+
run_locks.LEASE_TTL_S, run_locks.RENEW_EVERY_S, run_locks.LOCAL_VALIDITY_S = 2.0, 0.2, 1.0
|
|
559
|
+
chat.RECONCILE_INTERVAL_S, chat.RECONCILE_GRACE_S = 0.5, 0.0
|
|
560
|
+
a2a.TASK_SWEEP_INTERVAL_S = 0.5
|
|
561
|
+
@tool
|
|
562
|
+
def probe(query: str) -> str:
|
|
563
|
+
\"\"\"Test-only tool: reports what it was asked about.\"\"\"
|
|
564
|
+
if "slow" in query:
|
|
565
|
+
time.sleep(60)
|
|
566
|
+
return "probe reading for " + query
|
|
567
|
+
agent.graph = create_agent(
|
|
568
|
+
model=agent.get_model(), tools=[probe], system_prompt=agent.SYSTEM_PROMPT,
|
|
569
|
+
middleware=agent.middleware(), context_schema=agent.AgentContext, name="test-agent",
|
|
570
|
+
).with_config(dict(recursion_limit=agent.recursion_limit()))
|
|
571
|
+
import uvicorn
|
|
572
|
+
from {agent}.fast_api_app import app
|
|
573
|
+
uvicorn.run(app, host="127.0.0.1", port=int(sys.argv[1]), log_config=None)
|
|
574
|
+
"""
|
|
575
|
+
|
|
576
|
+
|
|
577
|
+
def _free_port() -> int:
|
|
578
|
+
with socket.socket() as s:
|
|
579
|
+
s.bind(("127.0.0.1", 0))
|
|
580
|
+
return s.getsockname()[1]
|
|
581
|
+
|
|
582
|
+
|
|
583
|
+
def _start_server(dsn: str, port: int, log: Path) -> subprocess.Popen[bytes]:
|
|
584
|
+
env = {**os.environ, "CHECKPOINTER": "postgres", "POSTGRES_DSN": dsn, "LOG_FORMAT": "text"}
|
|
585
|
+
process = subprocess.Popen(
|
|
586
|
+
[sys.executable, "-c", SERVER_SCRIPT.format(agent=AGENT_DIR), str(port)],
|
|
587
|
+
cwd=PROJECT,
|
|
588
|
+
env=env,
|
|
589
|
+
stdout=log.open("ab"),
|
|
590
|
+
stderr=subprocess.STDOUT,
|
|
591
|
+
)
|
|
592
|
+
deadline = time.monotonic() + 30
|
|
593
|
+
while time.monotonic() < deadline:
|
|
594
|
+
with contextlib.suppress(httpx.HTTPError):
|
|
595
|
+
if httpx.get(f"http://127.0.0.1:{port}/ready", timeout=1).status_code == 200:
|
|
596
|
+
return process
|
|
597
|
+
time.sleep(0.2)
|
|
598
|
+
process.kill()
|
|
599
|
+
raise AssertionError("the server did not get ready: " + log.read_text()[-2000:])
|
|
600
|
+
|
|
601
|
+
|
|
602
|
+
def test_a_crash_mid_tool_call_leaves_a_usable_thread_and_an_interrupted_run(
|
|
603
|
+
dsn: str, tmp_path: Path
|
|
604
|
+
) -> None:
|
|
605
|
+
port = _free_port()
|
|
606
|
+
base = f"http://127.0.0.1:{port}"
|
|
607
|
+
log = tmp_path / "server.log"
|
|
608
|
+
thread = f"crash-{uuid.uuid4().hex[:8]}"
|
|
609
|
+
server = _start_server(dsn, port, log)
|
|
610
|
+
try:
|
|
611
|
+
body = {"thread_id": thread, "message": "Run the probe for slow city"}
|
|
612
|
+
with contextlib.suppress(httpx.HTTPError):
|
|
613
|
+
with httpx.stream("POST", f"{base}/chat", json=body, headers=AUTH, timeout=30) as r:
|
|
614
|
+
for line in r.iter_lines():
|
|
615
|
+
if line.startswith("event: tool.call"):
|
|
616
|
+
time.sleep(1.0) # the model step's checkpoint lands
|
|
617
|
+
server.send_signal(signal.SIGKILL) # OOM kill, node loss, ...
|
|
618
|
+
server.wait()
|
|
619
|
+
break
|
|
620
|
+
server = _start_server(dsn, port, log)
|
|
621
|
+
crashed = httpx.get(f"{base}/threads/{thread}/messages", headers=AUTH).json()
|
|
622
|
+
assert [m["role"] for m in crashed] == ["user", "assistant"]
|
|
623
|
+
for message in ("hello", "hello again"):
|
|
624
|
+
deadline = time.monotonic() + 10
|
|
625
|
+
while True:
|
|
626
|
+
r = httpx.post(
|
|
627
|
+
f"{base}/chat", json={"thread_id": thread, "message": message}, headers=AUTH
|
|
628
|
+
)
|
|
629
|
+
# The dead process's lease holds the thread until it expires (2 s here).
|
|
630
|
+
if r.status_code != 409 or time.monotonic() > deadline:
|
|
631
|
+
break
|
|
632
|
+
time.sleep(0.3)
|
|
633
|
+
events = parse_sse(r.text)
|
|
634
|
+
assert events[-1][0] == "message.end", events
|
|
635
|
+
messages = httpx.get(f"{base}/threads/{thread}/messages", headers=AUTH).json()
|
|
636
|
+
roles = [m["role"] for m in messages]
|
|
637
|
+
# The open call got its result right after it, before the next turn.
|
|
638
|
+
assert roles == ["user", "assistant", "tool", "user", "assistant", "user", "assistant"]
|
|
639
|
+
assert messages[2]["is_error"] and "interrupted" in messages[2]["content"]
|
|
640
|
+
assert messages[2]["tool_call_id"] == messages[1]["tool_calls"][0]["id"]
|
|
641
|
+
# The killed run is on record, as interrupted, once its lease has expired.
|
|
642
|
+
deadline = time.monotonic() + 15
|
|
643
|
+
while True:
|
|
644
|
+
statuses = asyncio.run(
|
|
645
|
+
_sql(
|
|
646
|
+
dsn,
|
|
647
|
+
"SELECT status FROM runs WHERE thread_id = %s ORDER BY created_at",
|
|
648
|
+
(thread,),
|
|
649
|
+
)
|
|
650
|
+
)
|
|
651
|
+
if (statuses and statuses[0] == ("interrupted",)) or time.monotonic() > deadline:
|
|
652
|
+
break
|
|
653
|
+
time.sleep(0.5)
|
|
654
|
+
assert statuses == [("interrupted",), ("ok",), ("ok",)], statuses
|
|
655
|
+
metrics_text = httpx.get(f"{base}/metrics").text
|
|
656
|
+
assert 'agent_runs_total{status="interrupted"} 1.0' in metrics_text
|
|
657
|
+
finally:
|
|
658
|
+
server.terminate()
|
|
659
|
+
with contextlib.suppress(subprocess.TimeoutExpired):
|
|
660
|
+
server.wait(10)
|
|
661
|
+
if server.poll() is None:
|
|
662
|
+
server.kill()
|
|
663
|
+
|
|
664
|
+
|
|
665
|
+
# --- A2A tasks across replicas -----------------------------------------------------------------
|
|
666
|
+
|
|
667
|
+
A2A_PATH = f"/a2a/{AGENT_DIR}"
|
|
668
|
+
|
|
669
|
+
|
|
670
|
+
def _a2a(base: str, method: str, params: dict[str, Any]) -> dict[str, Any]:
|
|
671
|
+
"""One JSON-RPC call on a new connection (a load balancer may send each to any replica)."""
|
|
672
|
+
r = httpx.post(
|
|
673
|
+
f"{base}{A2A_PATH}",
|
|
674
|
+
json={"jsonrpc": "2.0", "id": uuid.uuid4().hex, "method": method, "params": params},
|
|
675
|
+
headers={**AUTH, "A2A-Version": "1.0"},
|
|
676
|
+
timeout=30,
|
|
677
|
+
)
|
|
678
|
+
assert r.status_code == 200, r.text
|
|
679
|
+
return r.json()
|
|
680
|
+
|
|
681
|
+
|
|
682
|
+
def _say(text: str, **configuration: Any) -> dict[str, Any]:
|
|
683
|
+
message = {"messageId": uuid.uuid4().hex, "role": "ROLE_USER", "parts": [{"text": text}]}
|
|
684
|
+
return {"message": message, "configuration": configuration}
|
|
685
|
+
|
|
686
|
+
|
|
687
|
+
def _stop(server: subprocess.Popen[bytes]) -> None:
|
|
688
|
+
server.terminate()
|
|
689
|
+
with contextlib.suppress(subprocess.TimeoutExpired):
|
|
690
|
+
server.wait(10)
|
|
691
|
+
if server.poll() is None:
|
|
692
|
+
server.kill()
|
|
693
|
+
|
|
694
|
+
|
|
695
|
+
def test_a2a_tasks_are_seen_by_every_replica_and_survive_a_crash(dsn: str, tmp_path: Path) -> None:
|
|
696
|
+
ports = [_free_port(), _free_port()]
|
|
697
|
+
a, b = (f"http://127.0.0.1:{port}" for port in ports)
|
|
698
|
+
log = tmp_path / "server.log"
|
|
699
|
+
servers = [_start_server(dsn, port, log) for port in ports]
|
|
700
|
+
try:
|
|
701
|
+
task = _a2a(a, "SendMessage", _say("hello"))["result"]["task"]
|
|
702
|
+
assert task["status"]["state"] == "TASK_STATE_COMPLETED", task
|
|
703
|
+
# Another replica has the task, its reply and its listing (-32001 there before).
|
|
704
|
+
seen = _a2a(b, "GetTask", {"id": task["id"]})["result"]
|
|
705
|
+
assert seen["id"] == task["id"] and seen["artifacts"] == task["artifacts"]
|
|
706
|
+
listed = _a2a(b, "ListTasks", {"contextId": task["contextId"]})["result"]
|
|
707
|
+
assert [t["id"] for t in listed["tasks"]] == [task["id"]]
|
|
708
|
+
# The replica that ran it is killed; its replacement still has the task.
|
|
709
|
+
servers[0].send_signal(signal.SIGKILL)
|
|
710
|
+
servers[0].wait()
|
|
711
|
+
servers[0] = _start_server(dsn, ports[0], log)
|
|
712
|
+
again = _a2a(a, "GetTask", {"id": task["id"]})["result"]
|
|
713
|
+
assert again["status"]["state"] == "TASK_STATE_COMPLETED"
|
|
714
|
+
# Deleting the conversation on one replica deletes its tasks for every replica.
|
|
715
|
+
assert httpx.delete(f"{b}/threads/{task['contextId']}", headers=AUTH).status_code == 204
|
|
716
|
+
assert _a2a(a, "GetTask", {"id": task["id"]})["error"]["code"] == -32001
|
|
717
|
+
finally:
|
|
718
|
+
for server in servers:
|
|
719
|
+
_stop(server)
|
|
720
|
+
|
|
721
|
+
|
|
722
|
+
def test_an_a2a_task_whose_replica_dies_is_failed_and_not_canceled_elsewhere(
|
|
723
|
+
dsn: str, tmp_path: Path
|
|
724
|
+
) -> None:
|
|
725
|
+
ports = [_free_port(), _free_port()]
|
|
726
|
+
a, b = (f"http://127.0.0.1:{port}" for port in ports)
|
|
727
|
+
log = tmp_path / "server.log"
|
|
728
|
+
servers = [_start_server(dsn, port, log) for port in ports]
|
|
729
|
+
try:
|
|
730
|
+
request = _say("Run the probe for slow city", returnImmediately=True)
|
|
731
|
+
task = _a2a(a, "SendMessage", request)["result"]["task"]
|
|
732
|
+
deadline = time.monotonic() + 10
|
|
733
|
+
while not asyncio.run(
|
|
734
|
+
_sql(
|
|
735
|
+
dsn,
|
|
736
|
+
"SELECT 1 FROM thread_locks WHERE thread_id = %s AND expires_at > now()",
|
|
737
|
+
(task["contextId"],),
|
|
738
|
+
)
|
|
739
|
+
):
|
|
740
|
+
assert time.monotonic() < deadline, "the run never took its thread"
|
|
741
|
+
time.sleep(0.1)
|
|
742
|
+
# The run is in replica a: a cancel reaching b is refused, not reported as done.
|
|
743
|
+
refused = _a2a(b, "CancelTask", {"id": task["id"]})
|
|
744
|
+
assert refused.get("error", {}).get("code") == -32002, refused
|
|
745
|
+
assert "another replica" in refused["error"]["message"]
|
|
746
|
+
# So is a subscription there: the run's events happen in a only.
|
|
747
|
+
subscribe = {"jsonrpc": "2.0", "id": "s", "method": "SubscribeToTask"}
|
|
748
|
+
r = httpx.post(
|
|
749
|
+
f"{b}{A2A_PATH}",
|
|
750
|
+
json={**subscribe, "params": {"id": task["id"]}},
|
|
751
|
+
headers={**AUTH, "A2A-Version": "1.0"},
|
|
752
|
+
timeout=10,
|
|
753
|
+
)
|
|
754
|
+
assert '"code":-32004' in r.text.replace(" ", ""), r.text
|
|
755
|
+
assert _a2a(b, "GetTask", {"id": task["id"]})["result"]["status"]["state"] == (
|
|
756
|
+
"TASK_STATE_WORKING"
|
|
757
|
+
)
|
|
758
|
+
servers[0].send_signal(signal.SIGKILL) # the replica running it dies mid-run
|
|
759
|
+
servers[0].wait()
|
|
760
|
+
deadline = time.monotonic() + 20
|
|
761
|
+
while True:
|
|
762
|
+
status = _a2a(b, "GetTask", {"id": task["id"]})["result"]["status"]
|
|
763
|
+
if status["state"] != "TASK_STATE_WORKING" or time.monotonic() > deadline:
|
|
764
|
+
break
|
|
765
|
+
time.sleep(0.3)
|
|
766
|
+
assert status["state"] == "TASK_STATE_FAILED", status
|
|
767
|
+
assert "Send the message again" in status["message"]["parts"][0]["text"]
|
|
768
|
+
finally:
|
|
769
|
+
for server in servers:
|
|
770
|
+
_stop(server)
|