graph-agents-cli 0.3.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- graph_agents_cli/__init__.py +26 -0
- graph_agents_cli/_api_policy.py +2145 -0
- graph_agents_cli/_approvals.py +400 -0
- graph_agents_cli/_build.py +186 -0
- graph_agents_cli/_build_info.json +7 -0
- graph_agents_cli/_chat_client.py +462 -0
- graph_agents_cli/_click.py +157 -0
- graph_agents_cli/_defaults.py +139 -0
- graph_agents_cli/_experiments.py +64 -0
- graph_agents_cli/_http.py +192 -0
- graph_agents_cli/_output.py +83 -0
- graph_agents_cli/_project.py +462 -0
- graph_agents_cli/_remote.py +220 -0
- graph_agents_cli/_response_schema.py +264 -0
- graph_agents_cli/_runner.py +319 -0
- graph_agents_cli/_skills_check.py +274 -0
- graph_agents_cli/_tools.py +189 -0
- graph_agents_cli/_trust.py +66 -0
- graph_agents_cli/api/__init__.py +15 -0
- graph_agents_cli/api/_changes.py +506 -0
- graph_agents_cli/api/_files.py +658 -0
- graph_agents_cli/api/cmd_api.py +2480 -0
- graph_agents_cli/deploy/__init__.py +15 -0
- graph_agents_cli/deploy/_config.py +171 -0
- graph_agents_cli/deploy/_image.py +128 -0
- graph_agents_cli/deploy/_kube.py +286 -0
- graph_agents_cli/deploy/_modes.py +234 -0
- graph_agents_cli/deploy/_preflight.py +370 -0
- graph_agents_cli/deploy/_values.py +168 -0
- graph_agents_cli/deploy/cmd_deploy.py +1866 -0
- graph_agents_cli/deploy/gitops.py +562 -0
- graph_agents_cli/deploy/local_load.py +273 -0
- graph_agents_cli/dev/__init__.py +13 -0
- graph_agents_cli/dev/cmd_build.py +131 -0
- graph_agents_cli/dev/cmd_install.py +78 -0
- graph_agents_cli/dev/cmd_lint.py +119 -0
- graph_agents_cli/dev/cmd_playground.py +297 -0
- graph_agents_cli/dev/policy_check.py +1287 -0
- graph_agents_cli/eval/__init__.py +22 -0
- graph_agents_cli/eval/_client.py +670 -0
- graph_agents_cli/eval/_common.py +177 -0
- graph_agents_cli/eval/_judge.py +168 -0
- graph_agents_cli/eval/_judge_runner.py +238 -0
- graph_agents_cli/eval/_paths.py +212 -0
- graph_agents_cli/eval/checks.py +581 -0
- graph_agents_cli/eval/cmd_analyze.py +278 -0
- graph_agents_cli/eval/cmd_compare.py +284 -0
- graph_agents_cli/eval/cmd_eval_group.py +80 -0
- graph_agents_cli/eval/cmd_generate.py +558 -0
- graph_agents_cli/eval/cmd_grade.py +466 -0
- graph_agents_cli/eval/cmd_metric.py +156 -0
- graph_agents_cli/eval/cmd_run.py +370 -0
- graph_agents_cli/eval/cmd_submit.py +400 -0
- graph_agents_cli/eval/config.py +435 -0
- graph_agents_cli/eval/dataset.py +350 -0
- graph_agents_cli/eval/gate.py +420 -0
- graph_agents_cli/eval/transcript.py +192 -0
- graph_agents_cli/extension/__init__.py +13 -0
- graph_agents_cli/extension/_compat.py +86 -0
- graph_agents_cli/extension/_loader.py +293 -0
- graph_agents_cli/extension/_manifest.py +135 -0
- graph_agents_cli/extension/_overrides.py +195 -0
- graph_agents_cli/extension/_paths.py +91 -0
- graph_agents_cli/extension/_refs.py +193 -0
- graph_agents_cli/extension/_resolver.py +453 -0
- graph_agents_cli/extension/_schema.py +106 -0
- graph_agents_cli/extension/_spec.py +253 -0
- graph_agents_cli/extension/_sync.py +102 -0
- graph_agents_cli/extension/_trust.py +58 -0
- graph_agents_cli/extension/cmd_extension_add.py +259 -0
- graph_agents_cli/extension/cmd_extension_group.py +57 -0
- graph_agents_cli/extension/cmd_extension_list.py +56 -0
- graph_agents_cli/extension/cmd_extension_remove.py +61 -0
- graph_agents_cli/extension/cmd_extension_update.py +195 -0
- graph_agents_cli/info/__init__.py +13 -0
- graph_agents_cli/info/cmd_info.py +222 -0
- graph_agents_cli/infra/__init__.py +15 -0
- graph_agents_cli/infra/checks.py +1169 -0
- graph_agents_cli/infra/cmd_infra.py +103 -0
- graph_agents_cli/main.py +591 -0
- graph_agents_cli/peer/__init__.py +15 -0
- graph_agents_cli/peer/_generate.py +254 -0
- graph_agents_cli/peer/cmd_peer.py +1151 -0
- graph_agents_cli/run/__init__.py +13 -0
- graph_agents_cli/run/_local_server.py +1157 -0
- graph_agents_cli/run/_signals.py +141 -0
- graph_agents_cli/run/cmd_approvals.py +530 -0
- graph_agents_cli/run/cmd_run.py +1421 -0
- graph_agents_cli/scaffold/__init__.py +19 -0
- graph_agents_cli/scaffold/agents/README.md +24 -0
- graph_agents_cli/scaffold/agents/empty_py/.template/templateconfig.yaml +22 -0
- graph_agents_cli/scaffold/agents/langgraph/.env.example +292 -0
- graph_agents_cli/scaffold/agents/langgraph/.template/templateconfig.yaml +28 -0
- graph_agents_cli/scaffold/agents/langgraph/Dockerfile +59 -0
- graph_agents_cli/scaffold/agents/langgraph/Dockerfile.langgraph-server +59 -0
- graph_agents_cli/scaffold/agents/langgraph/README.md +571 -0
- graph_agents_cli/scaffold/agents/langgraph/api-policy.yaml +60 -0
- graph_agents_cli/scaffold/agents/langgraph/app/__init__.py +20 -0
- graph_agents_cli/scaffold/agents/langgraph/app/agent.py +174 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/__init__.py +15 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/a2a.py +2162 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/a2a_client.py +1167 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/api_client.py +4220 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/approvals.py +1349 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/auth.py +1986 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/chat.py +2962 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/checkpointer.py +432 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/content.py +569 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/db.py +580 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/limits.py +203 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/metrics.py +231 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/middleware.py +361 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/model.py +611 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/playground.py +230 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/run_locks.py +459 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/structured.py +755 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/telemetry.py +681 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/threads.py +493 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/token_exchange.py +959 -0
- graph_agents_cli/scaffold/agents/langgraph/app/fast_api_app.py +770 -0
- graph_agents_cli/scaffold/agents/langgraph/app/policies/__init__.py +55 -0
- graph_agents_cli/scaffold/agents/langgraph/app/policies/custom.py +97 -0
- graph_agents_cli/scaffold/agents/langgraph/app/tools/__init__.py +46 -0
- graph_agents_cli/scaffold/agents/langgraph/app/tools/example_api.py +92 -0
- graph_agents_cli/scaffold/agents/langgraph/app/tools/weather.py +33 -0
- graph_agents_cli/scaffold/agents/langgraph/langgraph.json +14 -0
- graph_agents_cli/scaffold/agents/langgraph/pyproject.toml +78 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/conftest.py +376 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/eval/datasets/basic-dataset.json +53 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/eval/eval_config.yaml +32 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/approval_graph.py +137 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/fake_issuer.py +216 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/fake_openai.py +357 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_a2a_outcomes.py +569 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_a2a_relay.py +479 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_api_surface.py +812 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_approvals.py +1367 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_approvals_server.py +794 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_cross_actor.py +497 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_cross_actor_server.py +247 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_history_repair.py +278 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_model_apis.py +242 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_postgres.py +637 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_resilience_postgres.py +770 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_runtime_guardrails.py +854 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_server_e2e.py +340 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_server_runtime.py +989 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_structured_answers.py +584 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_structured_server.py +222 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_token_exchange_issuer.py +650 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/load_test/.results/.placeholder +0 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/load_test/README.md +22 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/load_test/conftest.py +21 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/load_test/load_test.py +81 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_a2a_client.py +824 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_a2a_scoping.py +724 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_api_client.py +1214 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_api_client_hardening.py +716 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_api_policy_rpc.py +767 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_approval_ledger.py +1536 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_fake_model.py +115 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_jwt_policy.py +991 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_limits.py +310 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_logging.py +148 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_logging_hardening.py +271 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_policy.py +378 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_resilience.py +610 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_server_auth.py +702 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_structured.py +673 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_telemetry.py +404 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_thread_listing.py +255 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_threads.py +268 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_token_exchange.py +1320 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_untrusted_content.py +393 -0
- graph_agents_cli/scaffold/agents/langgraph/uv-fastapi.lock +2084 -0
- graph_agents_cli/scaffold/agents/langgraph/uv-langgraph-server.lock +2106 -0
- graph_agents_cli/scaffold/agents/langgraph/{{cookiecutter.agent_guidance_filename}} +129 -0
- graph_agents_cli/scaffold/base_templates/_shared/graph-agents-cli-manifest.yaml +36 -0
- graph_agents_cli/scaffold/base_templates/python/.dockerignore +32 -0
- graph_agents_cli/scaffold/base_templates/python/.github/CODEOWNERS +30 -0
- graph_agents_cli/scaffold/base_templates/python/.github/agent.env +7 -0
- graph_agents_cli/scaffold/base_templates/python/.github/workflows/pr_checks.yaml +214 -0
- graph_agents_cli/scaffold/base_templates/python/.gitignore +209 -0
- graph_agents_cli/scaffold/base_templates/python/tests/unit/test_dummy.py +23 -0
- graph_agents_cli/scaffold/base_templates/python/{{cookiecutter.agent_guidance_filename}} +35 -0
- graph_agents_cli/scaffold/cmd_scaffold_group.py +49 -0
- graph_agents_cli/scaffold/commands/__init__.py +13 -0
- graph_agents_cli/scaffold/commands/create.py +1424 -0
- graph_agents_cli/scaffold/commands/enhance.py +1652 -0
- graph_agents_cli/scaffold/commands/upgrade.py +570 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/.github/agent.env +12 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/.github/workflows/promote-to-prod.yaml +371 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/.github/workflows/staging.yaml +450 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/argocd/application-dev.yaml +43 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/argocd/application-prod.yaml +41 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/argocd/application-staging.yaml +43 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/.helmignore +14 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/Chart.yaml +21 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/examples/networkpolicy.yaml +103 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/NOTES.txt +48 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/_helpers.tpl +189 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/certificate.yaml +15 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/configmap.yaml +10 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/deployment.yaml +199 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/hpa.yaml +22 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/httproute.yaml +30 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/ingress.yaml +39 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/networkpolicy.yaml +48 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/pdb.yaml +13 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/postgresql-secret.yaml +37 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/service.yaml +15 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/serviceaccount.yaml +13 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/servicemonitor.yaml +42 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values-dev.yaml +22 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values-prod.yaml +45 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values-staging.yaml +29 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values.yaml +396 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/tests/integration/test_chart.py +269 -0
- graph_agents_cli/scaffold/deployment_targets/none/README.md +5 -0
- graph_agents_cli/scaffold/deployment_targets/none/python/README.md +6 -0
- graph_agents_cli/scaffold/utils/__init__.py +13 -0
- graph_agents_cli/scaffold/utils/backup.py +212 -0
- graph_agents_cli/scaffold/utils/build_record.py +257 -0
- graph_agents_cli/scaffold/utils/cli_options.py +184 -0
- graph_agents_cli/scaffold/utils/fs.py +83 -0
- graph_agents_cli/scaffold/utils/generate_locks.py +214 -0
- graph_agents_cli/scaffold/utils/generation_metadata.py +88 -0
- graph_agents_cli/scaffold/utils/keyedit.py +768 -0
- graph_agents_cli/scaffold/utils/keymerge.py +537 -0
- graph_agents_cli/scaffold/utils/language.py +138 -0
- graph_agents_cli/scaffold/utils/lock_utils.py +94 -0
- graph_agents_cli/scaffold/utils/logging.py +77 -0
- graph_agents_cli/scaffold/utils/manifest.py +292 -0
- graph_agents_cli/scaffold/utils/merge.py +970 -0
- graph_agents_cli/scaffold/utils/merge3.py +216 -0
- graph_agents_cli/scaffold/utils/openapi_seed.py +199 -0
- graph_agents_cli/scaffold/utils/remote_template.py +376 -0
- graph_agents_cli/scaffold/utils/template.py +1352 -0
- graph_agents_cli/scaffold/utils/upgrade.py +894 -0
- graph_agents_cli/scaffold/utils/version.py +438 -0
- graph_agents_cli/secrets/__init__.py +15 -0
- graph_agents_cli/secrets/_apply.py +954 -0
- graph_agents_cli/secrets/_required.py +188 -0
- graph_agents_cli/secrets/cmd_secrets.py +211 -0
- graph_agents_cli/setup/__init__.py +13 -0
- graph_agents_cli/setup/_antigravity.py +221 -0
- graph_agents_cli/setup/cmd_auth.py +1030 -0
- graph_agents_cli/setup/cmd_dev_token.py +513 -0
- graph_agents_cli/setup/cmd_setup.py +428 -0
- graph_agents_cli/setup/cmd_update.py +140 -0
- graph_agents_cli/skills/__init__.py +13 -0
- graph_agents_cli/skills/_bundle.py +65 -0
- graph_agents_cli/skills/data/README.md +19 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/SKILL.md +357 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/references/github-settings.md +113 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/references/gitops.md +137 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/references/kubernetes.md +315 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/references/secrets.md +160 -0
- graph_agents_cli/skills/data/graph-agents-cli-eval/SKILL.md +303 -0
- graph_agents_cli/skills/data/graph-agents-cli-eval/references/dataset_schema.md +282 -0
- graph_agents_cli/skills/data/graph-agents-cli-eval/references/metrics-guide.md +143 -0
- graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/SKILL.md +659 -0
- graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/references/langchain-models.md +124 -0
- graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/references/langgraph.md +235 -0
- graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/references/template-contract.md +477 -0
- graph_agents_cli/skills/data/graph-agents-cli-observability/SKILL.md +231 -0
- graph_agents_cli/skills/data/graph-agents-cli-observability/references/langsmith.md +46 -0
- graph_agents_cli/skills/data/graph-agents-cli-observability/references/otel.md +59 -0
- graph_agents_cli/skills/data/graph-agents-cli-scaffold/SKILL.md +414 -0
- graph_agents_cli/skills/data/graph-agents-cli-scaffold/references/flags.md +134 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/SKILL.md +478 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/brainstorming.md +118 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/commands.md +419 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/extension.md +156 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/internals.md +67 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/spec-template.md +56 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/terminology.md +119 -0
- graph_agents_cli/system/__init__.py +15 -0
- graph_agents_cli/system/_apply.py +519 -0
- graph_agents_cli/system/_checks.py +1023 -0
- graph_agents_cli/system/_deploy.py +215 -0
- graph_agents_cli/system/_model.py +363 -0
- graph_agents_cli/system/_system.py +664 -0
- graph_agents_cli/system/_views.py +208 -0
- graph_agents_cli/system/cmd_system.py +423 -0
- graph_agents_cli-0.3.1.dist-info/METADATA +162 -0
- graph_agents_cli-0.3.1.dist-info/RECORD +291 -0
- graph_agents_cli-0.3.1.dist-info/WHEEL +4 -0
- graph_agents_cli-0.3.1.dist-info/entry_points.txt +2 -0
- graph_agents_cli-0.3.1.dist-info/licenses/LICENSE +201 -0
- graph_agents_cli-0.3.1.dist-info/licenses/NOTICE +19 -0
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
# Copyright 2026 Google LLC
|
|
2
|
+
# Modifications Copyright 2026 graph-agents-cli contributors
|
|
3
|
+
#
|
|
4
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
5
|
+
# you may not use this file except in compliance with the License.
|
|
6
|
+
# You may obtain a copy of the License at
|
|
7
|
+
#
|
|
8
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
9
|
+
#
|
|
10
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
11
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
12
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
13
|
+
# See the License for the specific language governing permissions and
|
|
14
|
+
# limitations under the License.
|
|
15
|
+
|
|
16
|
+
"""Evaluation commands: ``generate -> grade -> compare / analyze / submit``.
|
|
17
|
+
|
|
18
|
+
Nothing in this package imports a model SDK, LangChain, or LangGraph at import
|
|
19
|
+
time, so the CLI starts fast and installs light. Deterministic checks run in the CLI process; model
|
|
20
|
+
judges and custom metrics run inside the project's own environment through the
|
|
21
|
+
staged ``_judge_runner.py`` script (see ``_judge.py``).
|
|
22
|
+
"""
|
|
@@ -0,0 +1,670 @@
|
|
|
1
|
+
# Copyright 2026 graph-agents-cli contributors
|
|
2
|
+
#
|
|
3
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
4
|
+
# you may not use this file except in compliance with the License.
|
|
5
|
+
# You may obtain a copy of the License at
|
|
6
|
+
#
|
|
7
|
+
# https://www.apache.org/licenses/LICENSE-2.0
|
|
8
|
+
#
|
|
9
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
10
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
11
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
12
|
+
# See the License for the specific language governing permissions and
|
|
13
|
+
# limitations under the License.
|
|
14
|
+
|
|
15
|
+
"""Drive one eval case through ``POST /chat`` (the app's SSE chat API) into a trace.
|
|
16
|
+
|
|
17
|
+
The SSE transport is ``graph_agents_cli._chat_client.post_chat``; it is imported
|
|
18
|
+
lazily so tests can replace it. A trace (one JSON object per case) is derived
|
|
19
|
+
from the events: ``message.delta`` text becomes ``response``, ``tool.call`` /
|
|
20
|
+
``tool.result`` pairs become ``tool_calls``, ``message.end`` supplies ``usage``,
|
|
21
|
+
``latency_ms``, ``thread_id``, ``run_id`` and, for a project with a response
|
|
22
|
+
schema, ``structured_response`` (the run's JSON answer, which the
|
|
23
|
+
``json_schema`` check reads instead of the text); an ``error`` event or any
|
|
24
|
+
exception yields ``status: error``; a stream with no response yields
|
|
25
|
+
``status: missing``.
|
|
26
|
+
|
|
27
|
+
A run that pauses on a gated call (``message.end`` with ``status:
|
|
28
|
+
awaiting_approval``) is decided as the case's ``approvals`` instructions say
|
|
29
|
+
(the first whose ``match`` names the call), and the resumed run's events are
|
|
30
|
+
folded into the same turn. Every gate is recorded under ``approvals`` in the
|
|
31
|
+
turn (and the trace); a gate no instruction matches ends the case with
|
|
32
|
+
``status: error``: an unattended eval never approves anything on its own.
|
|
33
|
+
Nor does it leave an approval pending (which would keep the eval's thread
|
|
34
|
+
blocked, and leave a write an approver could still approve later): a gate the
|
|
35
|
+
case does not decide (no instruction, too many gates, a refused decision) is
|
|
36
|
+
rejected, as whoever may decide it. When the eval may not reject it (a
|
|
37
|
+
``role:`` gate without an approver credential, or with one that may not
|
|
38
|
+
decide it), the case's thread is deleted as the eval identity, which started
|
|
39
|
+
it and owns it: deleting a thread deletes its approvals. The record's
|
|
40
|
+
``cleanup`` says how that ended (``rejected``; ``not_pending`` when the
|
|
41
|
+
server says it no longer waits; ``thread_deleted``; ``left_pending``, with
|
|
42
|
+
its id in the case error, only when the thread could not be deleted either).
|
|
43
|
+
"""
|
|
44
|
+
|
|
45
|
+
from __future__ import annotations
|
|
46
|
+
|
|
47
|
+
import itertools
|
|
48
|
+
import time
|
|
49
|
+
from collections.abc import Iterable, Iterator, Mapping, Sequence
|
|
50
|
+
from dataclasses import replace
|
|
51
|
+
from typing import Any
|
|
52
|
+
|
|
53
|
+
from graph_agents_cli._approvals import Approval, call_matches, describe_match
|
|
54
|
+
from graph_agents_cli._chat_client import (
|
|
55
|
+
DECISION_APPROVE,
|
|
56
|
+
DECISION_REJECT,
|
|
57
|
+
STATUS_AWAITING_APPROVAL,
|
|
58
|
+
ChatHTTPError,
|
|
59
|
+
redact_credentials,
|
|
60
|
+
)
|
|
61
|
+
from graph_agents_cli.eval.dataset import EvalCase
|
|
62
|
+
|
|
63
|
+
STATUS_OK = "ok"
|
|
64
|
+
STATUS_ERROR = "error"
|
|
65
|
+
STATUS_MISSING = "missing"
|
|
66
|
+
|
|
67
|
+
DEFAULT_TIMEOUT = 300.0
|
|
68
|
+
|
|
69
|
+
# Gates one turn may pass: a run that keeps pausing is a case error, not a loop.
|
|
70
|
+
MAX_APPROVALS_PER_TURN = 20
|
|
71
|
+
|
|
72
|
+
# What a recorded gate ended as (trace `approvals[].status`).
|
|
73
|
+
APPROVAL_APPROVED = "approved"
|
|
74
|
+
APPROVAL_REJECTED = "rejected"
|
|
75
|
+
APPROVAL_UNEXPECTED = "unexpected" # no instruction matched: the case errors
|
|
76
|
+
# The server refused the decision: HTTP status -> recorded status.
|
|
77
|
+
_REFUSED_STATUS = {403: "forbidden", 404: "not_found", 409: "not_pending", 410: "expired"}
|
|
78
|
+
# How a gate the case did not decide was closed (trace `approvals[].cleanup`).
|
|
79
|
+
CLEANUP_REJECTED = "rejected" # the eval rejected it: nothing waits, nothing is sent
|
|
80
|
+
CLEANUP_NOT_PENDING = "not_pending" # the server says it no longer waits (decided, expired, gone)
|
|
81
|
+
# The eval may not reject it: the case's thread was deleted, and the approval with it.
|
|
82
|
+
CLEANUP_THREAD_DELETED = "thread_deleted"
|
|
83
|
+
# Nor could the thread be deleted: it waits until it expires or an approver decides it.
|
|
84
|
+
CLEANUP_LEFT_PENDING = "left_pending"
|
|
85
|
+
# The 409 code of a decision on an approval that is decided already.
|
|
86
|
+
_NOT_PENDING_CODE = "approval_not_pending"
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def build_case_headers(
|
|
90
|
+
header: tuple[str, ...],
|
|
91
|
+
cookie: tuple[str, ...],
|
|
92
|
+
session_token: str | None,
|
|
93
|
+
*,
|
|
94
|
+
env: Mapping[str, str] | None = None,
|
|
95
|
+
) -> dict[str, str]:
|
|
96
|
+
"""Headers for the chat calls, via ``graph_agents_cli._remote.build_headers``.
|
|
97
|
+
|
|
98
|
+
``env`` lets the local-server path fall back to the project's ``.env``
|
|
99
|
+
``API_KEY`` (as ``GRAPH_AGENTS_CLI_API_KEY``) when no bearer was given.
|
|
100
|
+
"""
|
|
101
|
+
from graph_agents_cli._remote import build_headers
|
|
102
|
+
|
|
103
|
+
return build_headers(header, cookie, session_token, env=env)
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def _stream(base_url: str, message: str, **kwargs: Any):
|
|
107
|
+
from graph_agents_cli import _chat_client
|
|
108
|
+
|
|
109
|
+
return _chat_client.post_chat(base_url, message, **kwargs)
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def _decide(base_url: str, thread_id: str, approval_id: str, decision: str, **kwargs: Any):
|
|
113
|
+
from graph_agents_cli import _chat_client
|
|
114
|
+
|
|
115
|
+
return _chat_client.decide_approval(base_url, thread_id, approval_id, decision, **kwargs)
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def _delete_thread(base_url: str, thread_id: str, **kwargs: Any) -> None:
|
|
119
|
+
from graph_agents_cli import _chat_client
|
|
120
|
+
|
|
121
|
+
_chat_client.delete_thread(base_url, thread_id, **kwargs)
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def _opened(events: Iterable[Any]) -> Iterator[Any]:
|
|
125
|
+
"""``events`` with the first one already read: an HTTP refusal raises here, not later."""
|
|
126
|
+
iterator = iter(events)
|
|
127
|
+
try:
|
|
128
|
+
first = next(iterator)
|
|
129
|
+
except StopIteration:
|
|
130
|
+
return iter(())
|
|
131
|
+
return itertools.chain([first], iterator)
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def find_instruction(
|
|
135
|
+
instructions: Sequence[Mapping[str, Any]], approval: Approval
|
|
136
|
+
) -> Mapping[str, Any] | None:
|
|
137
|
+
"""The first ``{decision, match}`` instruction whose match names the gated call."""
|
|
138
|
+
return next((i for i in instructions if call_matches(i["match"], approval)), None)
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def approval_record(approval: Approval) -> dict[str, Any]:
|
|
142
|
+
"""What a trace keeps of one gate: which call, who could decide, and how it ended."""
|
|
143
|
+
return {
|
|
144
|
+
"approval_id": approval.approval_id,
|
|
145
|
+
"api": approval.api,
|
|
146
|
+
"method": approval.method,
|
|
147
|
+
"path": approval.path,
|
|
148
|
+
"operation_id": approval.operation_id,
|
|
149
|
+
"approvers": list(approval.approvers),
|
|
150
|
+
# The instruction that decided it (its match, and approve or reject).
|
|
151
|
+
"match": None,
|
|
152
|
+
"decision": None,
|
|
153
|
+
"status": None,
|
|
154
|
+
"error": None,
|
|
155
|
+
# How a gate the case did not decide was closed; None when the case decided it.
|
|
156
|
+
"cleanup": None,
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def _describe_gate(approval: Approval) -> str:
|
|
161
|
+
call = f"{approval.method or '?'} {approval.path or '?'}"
|
|
162
|
+
extra = [f"api {approval.api}"] if approval.api else []
|
|
163
|
+
if approval.operation_id:
|
|
164
|
+
extra.append(f"operation_id {approval.operation_id}")
|
|
165
|
+
return f"{call} ({', '.join(extra)})" if extra else call
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def empty_trace(case_id: str) -> dict[str, Any]:
|
|
169
|
+
return {
|
|
170
|
+
"case_id": case_id,
|
|
171
|
+
"status": STATUS_MISSING,
|
|
172
|
+
"response": None,
|
|
173
|
+
"tool_calls": [],
|
|
174
|
+
"usage": None,
|
|
175
|
+
"latency_ms": None,
|
|
176
|
+
"error": None,
|
|
177
|
+
"thread_id": None,
|
|
178
|
+
"run_id": None,
|
|
179
|
+
"approvals": [],
|
|
180
|
+
"structured_response": None,
|
|
181
|
+
}
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def _add_usage(total: Any, usage: Any) -> Any:
|
|
185
|
+
"""Token usage summed over the runs of one turn (a paused run and its continuations)."""
|
|
186
|
+
if not isinstance(usage, dict):
|
|
187
|
+
return total
|
|
188
|
+
if not isinstance(total, dict):
|
|
189
|
+
return dict(usage)
|
|
190
|
+
summed = dict(total)
|
|
191
|
+
for key, value in usage.items():
|
|
192
|
+
if isinstance(value, int | float) and isinstance(summed.get(key), int | float):
|
|
193
|
+
summed[key] = summed[key] + value
|
|
194
|
+
else:
|
|
195
|
+
summed[key] = value
|
|
196
|
+
return summed
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
def _consume_turn(
|
|
200
|
+
base_url: str,
|
|
201
|
+
message: str,
|
|
202
|
+
*,
|
|
203
|
+
thread_id: str | None,
|
|
204
|
+
headers: Mapping[str, str],
|
|
205
|
+
metadata: Mapping[str, Any],
|
|
206
|
+
timeout: float,
|
|
207
|
+
approvals: Sequence[Mapping[str, Any]] = (),
|
|
208
|
+
decision_headers: Mapping[str, str] | None = None,
|
|
209
|
+
case_id: str | None = None,
|
|
210
|
+
) -> dict[str, Any]:
|
|
211
|
+
"""Send one user message and fold the stream into a per-turn record.
|
|
212
|
+
|
|
213
|
+
A pause on a gated call is decided per ``approvals`` and the
|
|
214
|
+
continuation is folded into the same record; latency and usage add up
|
|
215
|
+
over the runs. A gate that lists ``requester`` is decided as the eval
|
|
216
|
+
identity (``headers``: it started the run); any other with
|
|
217
|
+
``decision_headers`` (an approver's credential) when given.
|
|
218
|
+
"""
|
|
219
|
+
turn: dict[str, Any] = {
|
|
220
|
+
"status": STATUS_MISSING,
|
|
221
|
+
"response": None,
|
|
222
|
+
"tool_calls": [],
|
|
223
|
+
"usage": None,
|
|
224
|
+
"latency_ms": None,
|
|
225
|
+
"error": None,
|
|
226
|
+
"thread_id": thread_id,
|
|
227
|
+
"run_id": None,
|
|
228
|
+
"approvals": [],
|
|
229
|
+
"structured_response": None,
|
|
230
|
+
}
|
|
231
|
+
text_parts: list[str] = []
|
|
232
|
+
calls_by_id: dict[str, dict[str, Any]] = {}
|
|
233
|
+
saw_end = False
|
|
234
|
+
started = time.monotonic()
|
|
235
|
+
try:
|
|
236
|
+
events: Iterable[Any] = _stream(
|
|
237
|
+
base_url,
|
|
238
|
+
message,
|
|
239
|
+
thread_id=thread_id,
|
|
240
|
+
headers=dict(headers),
|
|
241
|
+
timeout=timeout,
|
|
242
|
+
metadata=dict(metadata),
|
|
243
|
+
)
|
|
244
|
+
while True:
|
|
245
|
+
end = _fold(events, turn, text_parts, calls_by_id)
|
|
246
|
+
if end is None:
|
|
247
|
+
break
|
|
248
|
+
saw_end = True
|
|
249
|
+
if end.get("status") != STATUS_AWAITING_APPROVAL:
|
|
250
|
+
break
|
|
251
|
+
saw_end = False
|
|
252
|
+
events = _resolve_gate(
|
|
253
|
+
base_url,
|
|
254
|
+
end,
|
|
255
|
+
turn,
|
|
256
|
+
approvals=approvals,
|
|
257
|
+
headers=headers,
|
|
258
|
+
approver_headers=decision_headers,
|
|
259
|
+
timeout=timeout,
|
|
260
|
+
case_id=case_id,
|
|
261
|
+
)
|
|
262
|
+
if events is None:
|
|
263
|
+
break
|
|
264
|
+
except Exception as exc: # transport, HTTP or protocol failure: record, do not abort
|
|
265
|
+
turn["status"] = STATUS_ERROR
|
|
266
|
+
# Printed and stored in the traces and results: never a URL's credentials.
|
|
267
|
+
turn["error"] = redact_credentials(f"{type(exc).__name__}: {exc}")
|
|
268
|
+
|
|
269
|
+
elapsed_ms = round((time.monotonic() - started) * 1000)
|
|
270
|
+
if turn["latency_ms"] is None:
|
|
271
|
+
turn["latency_ms"] = elapsed_ms
|
|
272
|
+
if text_parts:
|
|
273
|
+
turn["response"] = "".join(text_parts)
|
|
274
|
+
if turn["status"] == STATUS_ERROR:
|
|
275
|
+
return turn
|
|
276
|
+
if not saw_end:
|
|
277
|
+
if turn["response"] is None and not turn["tool_calls"]:
|
|
278
|
+
turn["status"] = STATUS_MISSING
|
|
279
|
+
turn["error"] = "stream produced no events"
|
|
280
|
+
else:
|
|
281
|
+
turn["status"] = STATUS_ERROR
|
|
282
|
+
turn["error"] = "stream closed before message.end"
|
|
283
|
+
return turn
|
|
284
|
+
if turn["response"] is None:
|
|
285
|
+
# A completed run with no text: keep it graded (an empty reply is a
|
|
286
|
+
# reply) rather than counting it as missing.
|
|
287
|
+
turn["response"] = ""
|
|
288
|
+
turn["status"] = STATUS_OK
|
|
289
|
+
return turn
|
|
290
|
+
|
|
291
|
+
|
|
292
|
+
def _resolve_gate(
|
|
293
|
+
base_url: str,
|
|
294
|
+
end: Mapping[str, Any],
|
|
295
|
+
turn: dict[str, Any],
|
|
296
|
+
*,
|
|
297
|
+
approvals: Sequence[Mapping[str, Any]],
|
|
298
|
+
headers: Mapping[str, str],
|
|
299
|
+
approver_headers: Mapping[str, str] | None = None,
|
|
300
|
+
timeout: float,
|
|
301
|
+
case_id: str | None,
|
|
302
|
+
) -> Iterator[Any] | None:
|
|
303
|
+
"""Decide the gate a run paused on; the resumed run's events, or None (turn errored).
|
|
304
|
+
|
|
305
|
+
As the eval identity (``headers``) when the gate lists ``requester``;
|
|
306
|
+
otherwise as the approver (``approver_headers``) when one is set.
|
|
307
|
+
"""
|
|
308
|
+
approval = Approval.from_payload(end.get("approval"), thread_id=turn["thread_id"])
|
|
309
|
+
if approval is not None and turn["thread_id"]:
|
|
310
|
+
# The decision is about this case's own thread, whatever the payload names.
|
|
311
|
+
approval = replace(approval, thread_id=turn["thread_id"])
|
|
312
|
+
if approval is None or not approval.thread_id:
|
|
313
|
+
turn["status"] = STATUS_ERROR
|
|
314
|
+
turn["error"] = "the run paused for an approval without an approval id or thread id"
|
|
315
|
+
return None
|
|
316
|
+
record = approval_record(approval)
|
|
317
|
+
turn["approvals"].append(record)
|
|
318
|
+
cleanup = {
|
|
319
|
+
"headers": headers,
|
|
320
|
+
"approver_headers": approver_headers,
|
|
321
|
+
"timeout": timeout,
|
|
322
|
+
"case_id": case_id,
|
|
323
|
+
}
|
|
324
|
+
if len(turn["approvals"]) > MAX_APPROVALS_PER_TURN:
|
|
325
|
+
record["status"] = APPROVAL_UNEXPECTED
|
|
326
|
+
turn["status"] = STATUS_ERROR
|
|
327
|
+
turn["error"] = f"the run paused on more than {MAX_APPROVALS_PER_TURN} gated calls"
|
|
328
|
+
_close_undecided(base_url, approval, record, turn, why="too many gated calls", **cleanup)
|
|
329
|
+
return None
|
|
330
|
+
instruction = find_instruction(approvals, approval)
|
|
331
|
+
if instruction is None:
|
|
332
|
+
record["status"] = APPROVAL_UNEXPECTED
|
|
333
|
+
turn["status"] = STATUS_ERROR
|
|
334
|
+
turn["error"] = (
|
|
335
|
+
f"unexpected approval gate on {_describe_gate(approval)}: the case has no "
|
|
336
|
+
"approvals instruction matching it; add one, e.g. "
|
|
337
|
+
'{"decision": "reject", "match": {...}}, saying what a human would decide'
|
|
338
|
+
)
|
|
339
|
+
_close_undecided(
|
|
340
|
+
base_url, approval, record, turn, why="no approvals instruction matches it", **cleanup
|
|
341
|
+
)
|
|
342
|
+
return None
|
|
343
|
+
decision = str(instruction["decision"])
|
|
344
|
+
record["decision"] = decision
|
|
345
|
+
record["match"] = describe_match(instruction["match"])
|
|
346
|
+
as_approver = approver_headers is not None and not approval.requester_may_decide
|
|
347
|
+
try:
|
|
348
|
+
events = _opened(
|
|
349
|
+
_decide(
|
|
350
|
+
base_url,
|
|
351
|
+
approval.thread_id,
|
|
352
|
+
approval.approval_id,
|
|
353
|
+
decision,
|
|
354
|
+
comment=f"eval case {case_id}" if case_id else "eval",
|
|
355
|
+
headers=dict(_decider_headers(approval, headers, approver_headers)),
|
|
356
|
+
timeout=timeout,
|
|
357
|
+
)
|
|
358
|
+
)
|
|
359
|
+
except ChatHTTPError as exc:
|
|
360
|
+
record["status"] = _REFUSED_STATUS.get(exc.status_code, "error")
|
|
361
|
+
record["error"] = f"HTTP {exc.status_code}"
|
|
362
|
+
turn["status"] = STATUS_ERROR
|
|
363
|
+
turn["error"] = redact_credentials(
|
|
364
|
+
f"the {decision} decision on {_describe_gate(approval)} was refused "
|
|
365
|
+
f"(HTTP {exc.status_code}: {exc.body[:200]})"
|
|
366
|
+
)
|
|
367
|
+
if exc.status_code == 403:
|
|
368
|
+
turn["error"] += (
|
|
369
|
+
"; the principal of GRAPH_AGENTS_CLI_APPROVER_API_KEY may not decide this gate "
|
|
370
|
+
"(it needs a role the gate lists, and must not be the eval identity)"
|
|
371
|
+
if as_approver
|
|
372
|
+
else "; the eval identity is not an approver of this gate (set "
|
|
373
|
+
"GRAPH_AGENTS_CLI_APPROVER_API_KEY to the credential of a principal holding a "
|
|
374
|
+
"role it lists)"
|
|
375
|
+
)
|
|
376
|
+
if _may_still_wait(exc):
|
|
377
|
+
_close_undecided(
|
|
378
|
+
base_url,
|
|
379
|
+
approval,
|
|
380
|
+
record,
|
|
381
|
+
turn,
|
|
382
|
+
why=f"its {decision} decision was refused",
|
|
383
|
+
# The same principal decides either way: a refused approve is a refused reject.
|
|
384
|
+
refused=exc if exc.status_code == 403 else None,
|
|
385
|
+
**cleanup,
|
|
386
|
+
)
|
|
387
|
+
else:
|
|
388
|
+
record["cleanup"] = CLEANUP_NOT_PENDING
|
|
389
|
+
return None
|
|
390
|
+
record["status"] = APPROVAL_APPROVED if decision == DECISION_APPROVE else APPROVAL_REJECTED
|
|
391
|
+
return events
|
|
392
|
+
|
|
393
|
+
|
|
394
|
+
def _decider_headers(
|
|
395
|
+
approval: Approval,
|
|
396
|
+
headers: Mapping[str, str],
|
|
397
|
+
approver_headers: Mapping[str, str] | None,
|
|
398
|
+
) -> Mapping[str, str]:
|
|
399
|
+
"""Who decides a gate: the eval identity when it lists ``requester`` (the eval
|
|
400
|
+
identity started the run), else the approver's credential when one is set."""
|
|
401
|
+
if approver_headers is not None and not approval.requester_may_decide:
|
|
402
|
+
return approver_headers
|
|
403
|
+
return headers
|
|
404
|
+
|
|
405
|
+
|
|
406
|
+
def _may_still_wait(exc: ChatHTTPError) -> bool:
|
|
407
|
+
"""Whether a refused decision can have left the approval pending."""
|
|
408
|
+
if exc.status_code in (404, 410): # gone, or expired (= rejected)
|
|
409
|
+
return False
|
|
410
|
+
return not (exc.status_code == 409 and _NOT_PENDING_CODE in (exc.body or ""))
|
|
411
|
+
|
|
412
|
+
|
|
413
|
+
def _close_undecided(
|
|
414
|
+
base_url: str,
|
|
415
|
+
approval: Approval,
|
|
416
|
+
record: dict[str, Any],
|
|
417
|
+
turn: dict[str, Any],
|
|
418
|
+
*,
|
|
419
|
+
why: str,
|
|
420
|
+
headers: Mapping[str, str],
|
|
421
|
+
approver_headers: Mapping[str, str] | None,
|
|
422
|
+
timeout: float,
|
|
423
|
+
case_id: str | None,
|
|
424
|
+
refused: ChatHTTPError | None = None,
|
|
425
|
+
) -> None:
|
|
426
|
+
"""Reject a gate the case does not decide, so the eval leaves no approval pending.
|
|
427
|
+
|
|
428
|
+
As whoever may decide it (``_decider_headers``). The resumed run is read
|
|
429
|
+
to its end, not folded into the turn: the case is an error already. A
|
|
430
|
+
gate that run pauses on in turn is rejected the same way (and recorded).
|
|
431
|
+
``refused``: the decision was refused as not allowed (403), so a reject
|
|
432
|
+
is too. A gate the eval could not reject goes with the case's thread
|
|
433
|
+
(``_delete_case_thread``). The outcome goes to each record's ``cleanup``
|
|
434
|
+
and to the case error.
|
|
435
|
+
"""
|
|
436
|
+
comment = f"eval case {case_id}: {why}" if case_id else f"eval: {why}"
|
|
437
|
+
current, current_record = approval, record
|
|
438
|
+
for _ in range(MAX_APPROVALS_PER_TURN):
|
|
439
|
+
events: Iterable[Any] = ()
|
|
440
|
+
if refused is None:
|
|
441
|
+
try:
|
|
442
|
+
events = _opened(
|
|
443
|
+
_decide(
|
|
444
|
+
base_url,
|
|
445
|
+
current.thread_id,
|
|
446
|
+
current.approval_id,
|
|
447
|
+
DECISION_REJECT,
|
|
448
|
+
comment=comment,
|
|
449
|
+
headers=dict(_decider_headers(current, headers, approver_headers)),
|
|
450
|
+
timeout=timeout,
|
|
451
|
+
)
|
|
452
|
+
)
|
|
453
|
+
except ChatHTTPError as exc:
|
|
454
|
+
refused = exc
|
|
455
|
+
except Exception as exc: # transport or protocol failure: it may still wait
|
|
456
|
+
_delete_case_thread(
|
|
457
|
+
base_url, current, current_record, turn, type(exc).__name__, headers, timeout
|
|
458
|
+
)
|
|
459
|
+
return
|
|
460
|
+
if refused is not None:
|
|
461
|
+
if _may_still_wait(refused):
|
|
462
|
+
_delete_case_thread(
|
|
463
|
+
base_url,
|
|
464
|
+
current,
|
|
465
|
+
current_record,
|
|
466
|
+
turn,
|
|
467
|
+
f"HTTP {refused.status_code}",
|
|
468
|
+
headers,
|
|
469
|
+
timeout,
|
|
470
|
+
)
|
|
471
|
+
else:
|
|
472
|
+
current_record["cleanup"] = CLEANUP_NOT_PENDING
|
|
473
|
+
return
|
|
474
|
+
current_record["cleanup"] = CLEANUP_REJECTED
|
|
475
|
+
try:
|
|
476
|
+
end = _drain(events)
|
|
477
|
+
except Exception: # the rejection holds; how its run ended is not the case's
|
|
478
|
+
return
|
|
479
|
+
following = (
|
|
480
|
+
Approval.from_payload(end.get("approval"), thread_id=current.thread_id)
|
|
481
|
+
if end is not None and end.get("status") == STATUS_AWAITING_APPROVAL
|
|
482
|
+
else None
|
|
483
|
+
)
|
|
484
|
+
if following is None:
|
|
485
|
+
return
|
|
486
|
+
# The rejected call's run paused on another gated call: close that one too.
|
|
487
|
+
current = replace(following, thread_id=current.thread_id)
|
|
488
|
+
current_record = approval_record(current)
|
|
489
|
+
current_record["status"] = APPROVAL_UNEXPECTED
|
|
490
|
+
turn["approvals"].append(current_record)
|
|
491
|
+
|
|
492
|
+
|
|
493
|
+
def _drain(events: Iterable[Any]) -> dict[str, Any] | None:
|
|
494
|
+
"""Read a resumed run to its end; its ``message.end`` data, or None."""
|
|
495
|
+
end = None
|
|
496
|
+
for event in events:
|
|
497
|
+
name = getattr(event, "event", None)
|
|
498
|
+
data = getattr(event, "data", None)
|
|
499
|
+
if name is None and isinstance(event, tuple):
|
|
500
|
+
name, data = event[0], event[1]
|
|
501
|
+
if name == "message.end" and isinstance(data, dict):
|
|
502
|
+
end = data
|
|
503
|
+
return end
|
|
504
|
+
|
|
505
|
+
|
|
506
|
+
def _delete_case_thread(
|
|
507
|
+
base_url: str,
|
|
508
|
+
approval: Approval,
|
|
509
|
+
record: dict[str, Any],
|
|
510
|
+
turn: dict[str, Any],
|
|
511
|
+
why: str,
|
|
512
|
+
headers: Mapping[str, str],
|
|
513
|
+
timeout: float,
|
|
514
|
+
) -> None:
|
|
515
|
+
"""Delete the case's thread, with the approval the eval could not reject.
|
|
516
|
+
|
|
517
|
+
As the eval identity (``headers``): it started the thread, so owns it,
|
|
518
|
+
and deleting a thread deletes its approvals, so nobody can approve the
|
|
519
|
+
eval's call later. Only when that fails too is the approval left pending
|
|
520
|
+
(named in the case error).
|
|
521
|
+
"""
|
|
522
|
+
gate = f"approval {approval.approval_id} on thread {approval.thread_id}"
|
|
523
|
+
try:
|
|
524
|
+
_delete_thread(base_url, approval.thread_id, headers=dict(headers), timeout=timeout)
|
|
525
|
+
except Exception as exc:
|
|
526
|
+
failure = (
|
|
527
|
+
f"HTTP {exc.status_code}" if isinstance(exc, ChatHTTPError) else type(exc).__name__
|
|
528
|
+
)
|
|
529
|
+
record["cleanup"] = CLEANUP_LEFT_PENDING
|
|
530
|
+
turn["error"] = (turn["error"] or "") + (
|
|
531
|
+
f"; {gate} is left pending (the eval could not reject it: {why}, nor delete "
|
|
532
|
+
f"its thread: {failure}): it expires on its own unless an approver approves "
|
|
533
|
+
"or rejects it first; reject it, or delete the thread"
|
|
534
|
+
)
|
|
535
|
+
return
|
|
536
|
+
record["cleanup"] = CLEANUP_THREAD_DELETED
|
|
537
|
+
turn["error"] = (turn["error"] or "") + (
|
|
538
|
+
f"; {gate} could not be rejected by the eval ({why}): the case's thread was "
|
|
539
|
+
"deleted, and the approval with it"
|
|
540
|
+
)
|
|
541
|
+
|
|
542
|
+
|
|
543
|
+
def _fold(
|
|
544
|
+
events: Iterable[Any],
|
|
545
|
+
turn: dict[str, Any],
|
|
546
|
+
text_parts: list[str],
|
|
547
|
+
calls_by_id: dict[str, dict[str, Any]],
|
|
548
|
+
) -> dict[str, Any] | None:
|
|
549
|
+
"""Fold one run's events into ``turn``; its ``message.end`` data, or None (error or no end)."""
|
|
550
|
+
for event in events:
|
|
551
|
+
name = getattr(event, "event", None)
|
|
552
|
+
data = getattr(event, "data", None)
|
|
553
|
+
if name is None and isinstance(event, tuple):
|
|
554
|
+
name, data = event[0], event[1]
|
|
555
|
+
data = data if isinstance(data, dict) else {}
|
|
556
|
+
if name == "message.start":
|
|
557
|
+
turn["thread_id"] = data.get("thread_id") or turn["thread_id"]
|
|
558
|
+
turn["run_id"] = data.get("run_id") or turn["run_id"]
|
|
559
|
+
elif name == "message.delta":
|
|
560
|
+
text_parts.append(str(data.get("text", "")))
|
|
561
|
+
elif name == "tool.call":
|
|
562
|
+
call = {
|
|
563
|
+
"id": data.get("id"),
|
|
564
|
+
"name": data.get("name"),
|
|
565
|
+
"args": data.get("args"),
|
|
566
|
+
"result": None,
|
|
567
|
+
"is_error": False,
|
|
568
|
+
}
|
|
569
|
+
turn["tool_calls"].append(call)
|
|
570
|
+
if call["id"] is not None:
|
|
571
|
+
calls_by_id[str(call["id"])] = call
|
|
572
|
+
elif name == "tool.result":
|
|
573
|
+
# A continuation's result belongs to the call its paused run made.
|
|
574
|
+
call = calls_by_id.get(str(data.get("id")))
|
|
575
|
+
if call is None:
|
|
576
|
+
call = {
|
|
577
|
+
"id": data.get("id"),
|
|
578
|
+
"name": data.get("name"),
|
|
579
|
+
"args": None,
|
|
580
|
+
"result": None,
|
|
581
|
+
"is_error": False,
|
|
582
|
+
}
|
|
583
|
+
turn["tool_calls"].append(call)
|
|
584
|
+
call["result"] = data.get("result")
|
|
585
|
+
call["is_error"] = bool(data.get("is_error", False))
|
|
586
|
+
elif name == "message.end":
|
|
587
|
+
turn["thread_id"] = data.get("thread_id") or turn["thread_id"]
|
|
588
|
+
turn["run_id"] = data.get("run_id") or turn["run_id"]
|
|
589
|
+
turn["usage"] = _add_usage(turn["usage"], data.get("usage"))
|
|
590
|
+
if "structured_response" in data:
|
|
591
|
+
# The run's answer in the project's response schema (the last run of the turn).
|
|
592
|
+
turn["structured_response"] = data["structured_response"]
|
|
593
|
+
latency = data.get("latency_ms")
|
|
594
|
+
if isinstance(latency, int | float):
|
|
595
|
+
previous = turn["latency_ms"]
|
|
596
|
+
turn["latency_ms"] = latency + (
|
|
597
|
+
previous if isinstance(previous, int | float) else 0
|
|
598
|
+
)
|
|
599
|
+
end_status = data.get("status", "ok")
|
|
600
|
+
if end_status not in ("ok", STATUS_AWAITING_APPROVAL):
|
|
601
|
+
turn["status"] = STATUS_ERROR
|
|
602
|
+
turn["error"] = f"message.end status {end_status!r}"
|
|
603
|
+
return None
|
|
604
|
+
return data
|
|
605
|
+
elif name == "error":
|
|
606
|
+
turn["status"] = STATUS_ERROR
|
|
607
|
+
code = data.get("code")
|
|
608
|
+
msg = data.get("message") or "unknown error"
|
|
609
|
+
turn["error"] = f"{msg} ({code})" if code else str(msg)
|
|
610
|
+
return None
|
|
611
|
+
return None
|
|
612
|
+
|
|
613
|
+
|
|
614
|
+
def run_case(
|
|
615
|
+
base_url: str,
|
|
616
|
+
case: EvalCase,
|
|
617
|
+
*,
|
|
618
|
+
headers: Mapping[str, str],
|
|
619
|
+
metadata: Mapping[str, Any] | None = None,
|
|
620
|
+
timeout: float = DEFAULT_TIMEOUT,
|
|
621
|
+
decision_headers: Mapping[str, str] | None = None,
|
|
622
|
+
) -> dict[str, Any]:
|
|
623
|
+
"""Run every user message of ``case`` on one thread and return its trace.
|
|
624
|
+
|
|
625
|
+
Multi-turn cases send each user message in order on the thread the first
|
|
626
|
+
``message.start`` reports; dataset ``assistant``/``system`` messages are
|
|
627
|
+
context for judges only and are not sent. The trace carries the final
|
|
628
|
+
turn's response, tool calls, usage and latency, plus ``turns`` with every
|
|
629
|
+
turn when there was more than one. Gated calls are decided per
|
|
630
|
+
``case.approvals`` (see ``_consume_turn``) and recorded under ``approvals``.
|
|
631
|
+
"""
|
|
632
|
+
trace = empty_trace(case.id)
|
|
633
|
+
turns: list[dict[str, Any]] = []
|
|
634
|
+
thread_id: str | None = None
|
|
635
|
+
base_metadata = {"source": "eval", "case_id": case.id, **(metadata or {})}
|
|
636
|
+
user_messages = case.user_messages()
|
|
637
|
+
for index, message in enumerate(user_messages):
|
|
638
|
+
turn = _consume_turn(
|
|
639
|
+
base_url,
|
|
640
|
+
message,
|
|
641
|
+
thread_id=thread_id,
|
|
642
|
+
headers=headers,
|
|
643
|
+
metadata={**base_metadata, "turn": index},
|
|
644
|
+
timeout=timeout,
|
|
645
|
+
approvals=case.approvals,
|
|
646
|
+
decision_headers=decision_headers,
|
|
647
|
+
case_id=case.id,
|
|
648
|
+
)
|
|
649
|
+
turns.append(turn)
|
|
650
|
+
thread_id = turn.get("thread_id") or thread_id
|
|
651
|
+
if turn["status"] != STATUS_OK:
|
|
652
|
+
break
|
|
653
|
+
last = turns[-1]
|
|
654
|
+
trace.update(
|
|
655
|
+
{
|
|
656
|
+
"status": last["status"],
|
|
657
|
+
"response": last["response"],
|
|
658
|
+
"tool_calls": last["tool_calls"],
|
|
659
|
+
"usage": last["usage"],
|
|
660
|
+
"latency_ms": last["latency_ms"],
|
|
661
|
+
"error": last["error"],
|
|
662
|
+
"thread_id": last["thread_id"] or thread_id,
|
|
663
|
+
"run_id": last["run_id"],
|
|
664
|
+
"approvals": last.get("approvals", []),
|
|
665
|
+
"structured_response": last.get("structured_response"),
|
|
666
|
+
}
|
|
667
|
+
)
|
|
668
|
+
if len(turns) > 1:
|
|
669
|
+
trace["turns"] = turns
|
|
670
|
+
return trace
|