graph-agents-cli 0.3.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- graph_agents_cli/__init__.py +26 -0
- graph_agents_cli/_api_policy.py +2145 -0
- graph_agents_cli/_approvals.py +400 -0
- graph_agents_cli/_build.py +186 -0
- graph_agents_cli/_build_info.json +7 -0
- graph_agents_cli/_chat_client.py +462 -0
- graph_agents_cli/_click.py +157 -0
- graph_agents_cli/_defaults.py +139 -0
- graph_agents_cli/_experiments.py +64 -0
- graph_agents_cli/_http.py +192 -0
- graph_agents_cli/_output.py +83 -0
- graph_agents_cli/_project.py +462 -0
- graph_agents_cli/_remote.py +220 -0
- graph_agents_cli/_response_schema.py +264 -0
- graph_agents_cli/_runner.py +319 -0
- graph_agents_cli/_skills_check.py +274 -0
- graph_agents_cli/_tools.py +189 -0
- graph_agents_cli/_trust.py +66 -0
- graph_agents_cli/api/__init__.py +15 -0
- graph_agents_cli/api/_changes.py +506 -0
- graph_agents_cli/api/_files.py +658 -0
- graph_agents_cli/api/cmd_api.py +2480 -0
- graph_agents_cli/deploy/__init__.py +15 -0
- graph_agents_cli/deploy/_config.py +171 -0
- graph_agents_cli/deploy/_image.py +128 -0
- graph_agents_cli/deploy/_kube.py +286 -0
- graph_agents_cli/deploy/_modes.py +234 -0
- graph_agents_cli/deploy/_preflight.py +370 -0
- graph_agents_cli/deploy/_values.py +168 -0
- graph_agents_cli/deploy/cmd_deploy.py +1866 -0
- graph_agents_cli/deploy/gitops.py +562 -0
- graph_agents_cli/deploy/local_load.py +273 -0
- graph_agents_cli/dev/__init__.py +13 -0
- graph_agents_cli/dev/cmd_build.py +131 -0
- graph_agents_cli/dev/cmd_install.py +78 -0
- graph_agents_cli/dev/cmd_lint.py +119 -0
- graph_agents_cli/dev/cmd_playground.py +297 -0
- graph_agents_cli/dev/policy_check.py +1287 -0
- graph_agents_cli/eval/__init__.py +22 -0
- graph_agents_cli/eval/_client.py +670 -0
- graph_agents_cli/eval/_common.py +177 -0
- graph_agents_cli/eval/_judge.py +168 -0
- graph_agents_cli/eval/_judge_runner.py +238 -0
- graph_agents_cli/eval/_paths.py +212 -0
- graph_agents_cli/eval/checks.py +581 -0
- graph_agents_cli/eval/cmd_analyze.py +278 -0
- graph_agents_cli/eval/cmd_compare.py +284 -0
- graph_agents_cli/eval/cmd_eval_group.py +80 -0
- graph_agents_cli/eval/cmd_generate.py +558 -0
- graph_agents_cli/eval/cmd_grade.py +466 -0
- graph_agents_cli/eval/cmd_metric.py +156 -0
- graph_agents_cli/eval/cmd_run.py +370 -0
- graph_agents_cli/eval/cmd_submit.py +400 -0
- graph_agents_cli/eval/config.py +435 -0
- graph_agents_cli/eval/dataset.py +350 -0
- graph_agents_cli/eval/gate.py +420 -0
- graph_agents_cli/eval/transcript.py +192 -0
- graph_agents_cli/extension/__init__.py +13 -0
- graph_agents_cli/extension/_compat.py +86 -0
- graph_agents_cli/extension/_loader.py +293 -0
- graph_agents_cli/extension/_manifest.py +135 -0
- graph_agents_cli/extension/_overrides.py +195 -0
- graph_agents_cli/extension/_paths.py +91 -0
- graph_agents_cli/extension/_refs.py +193 -0
- graph_agents_cli/extension/_resolver.py +453 -0
- graph_agents_cli/extension/_schema.py +106 -0
- graph_agents_cli/extension/_spec.py +253 -0
- graph_agents_cli/extension/_sync.py +102 -0
- graph_agents_cli/extension/_trust.py +58 -0
- graph_agents_cli/extension/cmd_extension_add.py +259 -0
- graph_agents_cli/extension/cmd_extension_group.py +57 -0
- graph_agents_cli/extension/cmd_extension_list.py +56 -0
- graph_agents_cli/extension/cmd_extension_remove.py +61 -0
- graph_agents_cli/extension/cmd_extension_update.py +195 -0
- graph_agents_cli/info/__init__.py +13 -0
- graph_agents_cli/info/cmd_info.py +222 -0
- graph_agents_cli/infra/__init__.py +15 -0
- graph_agents_cli/infra/checks.py +1169 -0
- graph_agents_cli/infra/cmd_infra.py +103 -0
- graph_agents_cli/main.py +591 -0
- graph_agents_cli/peer/__init__.py +15 -0
- graph_agents_cli/peer/_generate.py +254 -0
- graph_agents_cli/peer/cmd_peer.py +1151 -0
- graph_agents_cli/run/__init__.py +13 -0
- graph_agents_cli/run/_local_server.py +1157 -0
- graph_agents_cli/run/_signals.py +141 -0
- graph_agents_cli/run/cmd_approvals.py +530 -0
- graph_agents_cli/run/cmd_run.py +1421 -0
- graph_agents_cli/scaffold/__init__.py +19 -0
- graph_agents_cli/scaffold/agents/README.md +24 -0
- graph_agents_cli/scaffold/agents/empty_py/.template/templateconfig.yaml +22 -0
- graph_agents_cli/scaffold/agents/langgraph/.env.example +292 -0
- graph_agents_cli/scaffold/agents/langgraph/.template/templateconfig.yaml +28 -0
- graph_agents_cli/scaffold/agents/langgraph/Dockerfile +59 -0
- graph_agents_cli/scaffold/agents/langgraph/Dockerfile.langgraph-server +59 -0
- graph_agents_cli/scaffold/agents/langgraph/README.md +571 -0
- graph_agents_cli/scaffold/agents/langgraph/api-policy.yaml +60 -0
- graph_agents_cli/scaffold/agents/langgraph/app/__init__.py +20 -0
- graph_agents_cli/scaffold/agents/langgraph/app/agent.py +174 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/__init__.py +15 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/a2a.py +2162 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/a2a_client.py +1167 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/api_client.py +4220 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/approvals.py +1349 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/auth.py +1986 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/chat.py +2962 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/checkpointer.py +432 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/content.py +569 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/db.py +580 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/limits.py +203 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/metrics.py +231 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/middleware.py +361 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/model.py +611 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/playground.py +230 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/run_locks.py +459 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/structured.py +755 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/telemetry.py +681 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/threads.py +493 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/token_exchange.py +959 -0
- graph_agents_cli/scaffold/agents/langgraph/app/fast_api_app.py +770 -0
- graph_agents_cli/scaffold/agents/langgraph/app/policies/__init__.py +55 -0
- graph_agents_cli/scaffold/agents/langgraph/app/policies/custom.py +97 -0
- graph_agents_cli/scaffold/agents/langgraph/app/tools/__init__.py +46 -0
- graph_agents_cli/scaffold/agents/langgraph/app/tools/example_api.py +92 -0
- graph_agents_cli/scaffold/agents/langgraph/app/tools/weather.py +33 -0
- graph_agents_cli/scaffold/agents/langgraph/langgraph.json +14 -0
- graph_agents_cli/scaffold/agents/langgraph/pyproject.toml +78 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/conftest.py +376 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/eval/datasets/basic-dataset.json +53 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/eval/eval_config.yaml +32 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/approval_graph.py +137 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/fake_issuer.py +216 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/fake_openai.py +357 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_a2a_outcomes.py +569 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_a2a_relay.py +479 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_api_surface.py +812 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_approvals.py +1367 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_approvals_server.py +794 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_cross_actor.py +497 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_cross_actor_server.py +247 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_history_repair.py +278 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_model_apis.py +242 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_postgres.py +637 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_resilience_postgres.py +770 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_runtime_guardrails.py +854 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_server_e2e.py +340 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_server_runtime.py +989 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_structured_answers.py +584 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_structured_server.py +222 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_token_exchange_issuer.py +650 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/load_test/.results/.placeholder +0 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/load_test/README.md +22 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/load_test/conftest.py +21 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/load_test/load_test.py +81 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_a2a_client.py +824 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_a2a_scoping.py +724 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_api_client.py +1214 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_api_client_hardening.py +716 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_api_policy_rpc.py +767 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_approval_ledger.py +1536 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_fake_model.py +115 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_jwt_policy.py +991 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_limits.py +310 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_logging.py +148 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_logging_hardening.py +271 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_policy.py +378 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_resilience.py +610 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_server_auth.py +702 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_structured.py +673 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_telemetry.py +404 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_thread_listing.py +255 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_threads.py +268 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_token_exchange.py +1320 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_untrusted_content.py +393 -0
- graph_agents_cli/scaffold/agents/langgraph/uv-fastapi.lock +2084 -0
- graph_agents_cli/scaffold/agents/langgraph/uv-langgraph-server.lock +2106 -0
- graph_agents_cli/scaffold/agents/langgraph/{{cookiecutter.agent_guidance_filename}} +129 -0
- graph_agents_cli/scaffold/base_templates/_shared/graph-agents-cli-manifest.yaml +36 -0
- graph_agents_cli/scaffold/base_templates/python/.dockerignore +32 -0
- graph_agents_cli/scaffold/base_templates/python/.github/CODEOWNERS +30 -0
- graph_agents_cli/scaffold/base_templates/python/.github/agent.env +7 -0
- graph_agents_cli/scaffold/base_templates/python/.github/workflows/pr_checks.yaml +214 -0
- graph_agents_cli/scaffold/base_templates/python/.gitignore +209 -0
- graph_agents_cli/scaffold/base_templates/python/tests/unit/test_dummy.py +23 -0
- graph_agents_cli/scaffold/base_templates/python/{{cookiecutter.agent_guidance_filename}} +35 -0
- graph_agents_cli/scaffold/cmd_scaffold_group.py +49 -0
- graph_agents_cli/scaffold/commands/__init__.py +13 -0
- graph_agents_cli/scaffold/commands/create.py +1424 -0
- graph_agents_cli/scaffold/commands/enhance.py +1652 -0
- graph_agents_cli/scaffold/commands/upgrade.py +570 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/.github/agent.env +12 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/.github/workflows/promote-to-prod.yaml +371 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/.github/workflows/staging.yaml +450 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/argocd/application-dev.yaml +43 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/argocd/application-prod.yaml +41 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/argocd/application-staging.yaml +43 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/.helmignore +14 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/Chart.yaml +21 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/examples/networkpolicy.yaml +103 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/NOTES.txt +48 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/_helpers.tpl +189 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/certificate.yaml +15 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/configmap.yaml +10 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/deployment.yaml +199 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/hpa.yaml +22 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/httproute.yaml +30 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/ingress.yaml +39 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/networkpolicy.yaml +48 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/pdb.yaml +13 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/postgresql-secret.yaml +37 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/service.yaml +15 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/serviceaccount.yaml +13 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/servicemonitor.yaml +42 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values-dev.yaml +22 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values-prod.yaml +45 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values-staging.yaml +29 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values.yaml +396 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/tests/integration/test_chart.py +269 -0
- graph_agents_cli/scaffold/deployment_targets/none/README.md +5 -0
- graph_agents_cli/scaffold/deployment_targets/none/python/README.md +6 -0
- graph_agents_cli/scaffold/utils/__init__.py +13 -0
- graph_agents_cli/scaffold/utils/backup.py +212 -0
- graph_agents_cli/scaffold/utils/build_record.py +257 -0
- graph_agents_cli/scaffold/utils/cli_options.py +184 -0
- graph_agents_cli/scaffold/utils/fs.py +83 -0
- graph_agents_cli/scaffold/utils/generate_locks.py +214 -0
- graph_agents_cli/scaffold/utils/generation_metadata.py +88 -0
- graph_agents_cli/scaffold/utils/keyedit.py +768 -0
- graph_agents_cli/scaffold/utils/keymerge.py +537 -0
- graph_agents_cli/scaffold/utils/language.py +138 -0
- graph_agents_cli/scaffold/utils/lock_utils.py +94 -0
- graph_agents_cli/scaffold/utils/logging.py +77 -0
- graph_agents_cli/scaffold/utils/manifest.py +292 -0
- graph_agents_cli/scaffold/utils/merge.py +970 -0
- graph_agents_cli/scaffold/utils/merge3.py +216 -0
- graph_agents_cli/scaffold/utils/openapi_seed.py +199 -0
- graph_agents_cli/scaffold/utils/remote_template.py +376 -0
- graph_agents_cli/scaffold/utils/template.py +1352 -0
- graph_agents_cli/scaffold/utils/upgrade.py +894 -0
- graph_agents_cli/scaffold/utils/version.py +438 -0
- graph_agents_cli/secrets/__init__.py +15 -0
- graph_agents_cli/secrets/_apply.py +954 -0
- graph_agents_cli/secrets/_required.py +188 -0
- graph_agents_cli/secrets/cmd_secrets.py +211 -0
- graph_agents_cli/setup/__init__.py +13 -0
- graph_agents_cli/setup/_antigravity.py +221 -0
- graph_agents_cli/setup/cmd_auth.py +1030 -0
- graph_agents_cli/setup/cmd_dev_token.py +513 -0
- graph_agents_cli/setup/cmd_setup.py +428 -0
- graph_agents_cli/setup/cmd_update.py +140 -0
- graph_agents_cli/skills/__init__.py +13 -0
- graph_agents_cli/skills/_bundle.py +65 -0
- graph_agents_cli/skills/data/README.md +19 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/SKILL.md +357 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/references/github-settings.md +113 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/references/gitops.md +137 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/references/kubernetes.md +315 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/references/secrets.md +160 -0
- graph_agents_cli/skills/data/graph-agents-cli-eval/SKILL.md +303 -0
- graph_agents_cli/skills/data/graph-agents-cli-eval/references/dataset_schema.md +282 -0
- graph_agents_cli/skills/data/graph-agents-cli-eval/references/metrics-guide.md +143 -0
- graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/SKILL.md +659 -0
- graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/references/langchain-models.md +124 -0
- graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/references/langgraph.md +235 -0
- graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/references/template-contract.md +477 -0
- graph_agents_cli/skills/data/graph-agents-cli-observability/SKILL.md +231 -0
- graph_agents_cli/skills/data/graph-agents-cli-observability/references/langsmith.md +46 -0
- graph_agents_cli/skills/data/graph-agents-cli-observability/references/otel.md +59 -0
- graph_agents_cli/skills/data/graph-agents-cli-scaffold/SKILL.md +414 -0
- graph_agents_cli/skills/data/graph-agents-cli-scaffold/references/flags.md +134 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/SKILL.md +478 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/brainstorming.md +118 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/commands.md +419 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/extension.md +156 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/internals.md +67 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/spec-template.md +56 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/terminology.md +119 -0
- graph_agents_cli/system/__init__.py +15 -0
- graph_agents_cli/system/_apply.py +519 -0
- graph_agents_cli/system/_checks.py +1023 -0
- graph_agents_cli/system/_deploy.py +215 -0
- graph_agents_cli/system/_model.py +363 -0
- graph_agents_cli/system/_system.py +664 -0
- graph_agents_cli/system/_views.py +208 -0
- graph_agents_cli/system/cmd_system.py +423 -0
- graph_agents_cli-0.3.1.dist-info/METADATA +162 -0
- graph_agents_cli-0.3.1.dist-info/RECORD +291 -0
- graph_agents_cli-0.3.1.dist-info/WHEEL +4 -0
- graph_agents_cli-0.3.1.dist-info/entry_points.txt +2 -0
- graph_agents_cli-0.3.1.dist-info/licenses/LICENSE +201 -0
- graph_agents_cli-0.3.1.dist-info/licenses/NOTICE +19 -0
|
@@ -0,0 +1,278 @@
|
|
|
1
|
+
# Copyright 2026 graph-agents-cli contributors
|
|
2
|
+
#
|
|
3
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
4
|
+
# you may not use this file except in compliance with the License.
|
|
5
|
+
# You may obtain a copy of the License at
|
|
6
|
+
#
|
|
7
|
+
# https://www.apache.org/licenses/LICENSE-2.0
|
|
8
|
+
#
|
|
9
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
10
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
11
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
12
|
+
# See the License for the specific language governing permissions and
|
|
13
|
+
# limitations under the License.
|
|
14
|
+
|
|
15
|
+
"""graph-agents-cli eval analyze command: cluster failures by reason."""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import re
|
|
20
|
+
from pathlib import Path
|
|
21
|
+
from typing import Any
|
|
22
|
+
|
|
23
|
+
import click
|
|
24
|
+
from rich.table import Table
|
|
25
|
+
|
|
26
|
+
from graph_agents_cli._output import Console
|
|
27
|
+
from graph_agents_cli._project import find_project_root
|
|
28
|
+
from graph_agents_cli.eval import _paths
|
|
29
|
+
from graph_agents_cli.eval._common import (
|
|
30
|
+
EvalConfigError,
|
|
31
|
+
load_json_file,
|
|
32
|
+
resolve_judge_identity,
|
|
33
|
+
utc_now_iso,
|
|
34
|
+
write_json_file,
|
|
35
|
+
)
|
|
36
|
+
from graph_agents_cli.eval._judge import results_by_id, run_judge_runner
|
|
37
|
+
from graph_agents_cli.eval.config import load_eval_config
|
|
38
|
+
from graph_agents_cli.eval.gate import STATUS_PASSED, STATUS_RANK
|
|
39
|
+
|
|
40
|
+
_QUOTED = re.compile(r"'[^']*'|\"[^\"]*\"")
|
|
41
|
+
_NUMBER = re.compile(r"-?\d+(?:\.\d+)?")
|
|
42
|
+
_SPACES = re.compile(r"\s+")
|
|
43
|
+
_PATTERN_LIMIT = 120
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def normalize_reason(reason: str) -> tuple[str, str]:
|
|
47
|
+
"""``(kind, pattern)`` for a reason line: the prefix before ':' and a masked message.
|
|
48
|
+
|
|
49
|
+
Quoted literals become ``<x>`` and numbers ``#`` so ``contains: response does
|
|
50
|
+
not contain 'hello'`` and ``... 'goodbye'`` fall into one cluster.
|
|
51
|
+
"""
|
|
52
|
+
kind, sep, rest = reason.partition(":")
|
|
53
|
+
if not sep:
|
|
54
|
+
kind, rest = "other", reason
|
|
55
|
+
kind = kind.strip().lower() or "other"
|
|
56
|
+
if kind == "error":
|
|
57
|
+
# "error: judge <metric>: ..." / "error: custom metric <metric>: ..." -> keep
|
|
58
|
+
# the metric name out of the pattern.
|
|
59
|
+
sub_kind, sub_sep, sub_rest = rest.strip().partition(":")
|
|
60
|
+
if sub_sep and sub_kind.lower().startswith("judge "):
|
|
61
|
+
kind = "error/judge"
|
|
62
|
+
rest = sub_rest
|
|
63
|
+
elif sub_sep and sub_kind.lower().startswith("custom metric "):
|
|
64
|
+
kind = "error/custom"
|
|
65
|
+
rest = sub_rest
|
|
66
|
+
pattern = rest.strip().lower()
|
|
67
|
+
pattern = _QUOTED.sub("<x>", pattern)
|
|
68
|
+
pattern = _NUMBER.sub("#", pattern)
|
|
69
|
+
pattern = _SPACES.sub(" ", pattern).strip()
|
|
70
|
+
if len(pattern) > _PATTERN_LIMIT:
|
|
71
|
+
pattern = pattern[: _PATTERN_LIMIT - 3] + "..."
|
|
72
|
+
return kind, pattern
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def cluster_results(results: dict[str, Any]) -> list[dict[str, Any]]:
|
|
76
|
+
"""Deterministic clusters of every non-passed case's reasons."""
|
|
77
|
+
clusters: dict[tuple[str, str, str], dict[str, Any]] = {}
|
|
78
|
+
for case in results.get("cases", []):
|
|
79
|
+
status = case.get("status")
|
|
80
|
+
if status == STATUS_PASSED:
|
|
81
|
+
continue
|
|
82
|
+
reasons = case.get("reasons") or [f"{status}: no reason recorded"]
|
|
83
|
+
for reason in reasons:
|
|
84
|
+
kind, pattern = normalize_reason(str(reason))
|
|
85
|
+
key = (str(status), kind, pattern)
|
|
86
|
+
cluster = clusters.setdefault(
|
|
87
|
+
key,
|
|
88
|
+
{
|
|
89
|
+
"status": status,
|
|
90
|
+
"kind": kind,
|
|
91
|
+
"pattern": pattern,
|
|
92
|
+
"count": 0,
|
|
93
|
+
"case_ids": [],
|
|
94
|
+
"examples": [],
|
|
95
|
+
},
|
|
96
|
+
)
|
|
97
|
+
cluster["count"] += 1
|
|
98
|
+
if case["id"] not in cluster["case_ids"]:
|
|
99
|
+
cluster["case_ids"].append(case["id"])
|
|
100
|
+
if len(cluster["examples"]) < 3:
|
|
101
|
+
cluster["examples"].append(str(reason))
|
|
102
|
+
ordered = sorted(
|
|
103
|
+
clusters.values(),
|
|
104
|
+
key=lambda c: (-c["count"], -STATUS_RANK.get(c["status"], 0), c["kind"], c["pattern"]),
|
|
105
|
+
)
|
|
106
|
+
for index, cluster in enumerate(ordered, start=1):
|
|
107
|
+
cluster["rank"] = index
|
|
108
|
+
return ordered
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def _summary_prompt(results: dict[str, Any], clusters: list[dict[str, Any]]) -> str:
|
|
112
|
+
lines = [
|
|
113
|
+
"You are reviewing the failures of an automated evaluation of an AI agent.",
|
|
114
|
+
f"Summary: {results.get('summary')}",
|
|
115
|
+
"Failure clusters (status / check-or-metric / pattern / count / examples):",
|
|
116
|
+
]
|
|
117
|
+
for cluster in clusters[:20]:
|
|
118
|
+
lines.append(
|
|
119
|
+
f"- [{cluster['rank']}] {cluster['status']} / {cluster['kind']} / {cluster['pattern']} "
|
|
120
|
+
f"/ {cluster['count']} case(s): {cluster['case_ids'][:5]}"
|
|
121
|
+
)
|
|
122
|
+
for example in cluster["examples"]:
|
|
123
|
+
lines.append(f" e.g. {example}")
|
|
124
|
+
lines.append(
|
|
125
|
+
"For each cluster give the most likely root cause (prompt, tool, data, or expectation) "
|
|
126
|
+
"and one concrete fix. Finish with the three highest-priority actions. Be brief."
|
|
127
|
+
)
|
|
128
|
+
return "\n".join(lines)
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def _resolve_results_path(project_root: Path | None, results: str | None) -> Path:
|
|
132
|
+
if results:
|
|
133
|
+
path = Path(results)
|
|
134
|
+
if not path.is_absolute() and project_root is not None and not path.exists():
|
|
135
|
+
path = project_root / path
|
|
136
|
+
if not path.is_file():
|
|
137
|
+
raise EvalConfigError(f"results file not found: {path}")
|
|
138
|
+
return path
|
|
139
|
+
if project_root is None:
|
|
140
|
+
raise EvalConfigError("not inside a graph-agents-cli project; pass --results PATH")
|
|
141
|
+
latest = _paths.latest_file(
|
|
142
|
+
_paths.default_grade_results_dir(project_root), _paths.RESULTS_FILE_PREFIX
|
|
143
|
+
)
|
|
144
|
+
if latest is None:
|
|
145
|
+
raise EvalConfigError(
|
|
146
|
+
f"no results found under {_paths.default_grade_results_dir(project_root)}; "
|
|
147
|
+
"run `graph-agents-cli eval grade` first or pass --results PATH"
|
|
148
|
+
)
|
|
149
|
+
return latest
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
@click.command("analyze")
|
|
153
|
+
@click.option(
|
|
154
|
+
"--results",
|
|
155
|
+
default=None,
|
|
156
|
+
help="Results file. Defaults to the newest artifacts/grade_results/results_<ts>.json.",
|
|
157
|
+
)
|
|
158
|
+
@click.option(
|
|
159
|
+
"--output",
|
|
160
|
+
"output_path",
|
|
161
|
+
default=None,
|
|
162
|
+
help="Analysis file. Defaults to artifacts/analysis_<ts>.json.",
|
|
163
|
+
)
|
|
164
|
+
@click.option(
|
|
165
|
+
"--top-k", type=click.IntRange(min=1), default=None, help="Print only the K largest clusters."
|
|
166
|
+
)
|
|
167
|
+
@click.option(
|
|
168
|
+
"--judge", is_flag=True, help="Ask the judge model for root causes and fixes per cluster."
|
|
169
|
+
)
|
|
170
|
+
@click.option("--judge-provider", default=None, help="Override the judge provider (with --judge).")
|
|
171
|
+
@click.option("--judge-model", default=None, help="Override the judge model (with --judge).")
|
|
172
|
+
def cmd_analyze(
|
|
173
|
+
*,
|
|
174
|
+
results: str | None,
|
|
175
|
+
output_path: str | None,
|
|
176
|
+
top_k: int | None,
|
|
177
|
+
judge: bool,
|
|
178
|
+
judge_provider: str | None,
|
|
179
|
+
judge_model: str | None,
|
|
180
|
+
) -> None:
|
|
181
|
+
"""Cluster failed, quality-below-threshold, error and missing cases by reason.
|
|
182
|
+
|
|
183
|
+
Grouping is deterministic (status, check or metric, masked message), so two
|
|
184
|
+
runs of the same results file produce the same clusters. With --judge the
|
|
185
|
+
clusters are also summarised by the judge model through the project's
|
|
186
|
+
judge runner. The analysis is written to artifacts/analysis_<ts>.json.
|
|
187
|
+
"""
|
|
188
|
+
console = Console()
|
|
189
|
+
project_root = find_project_root()
|
|
190
|
+
results_path = _resolve_results_path(project_root, results)
|
|
191
|
+
data = load_json_file(results_path, "results file")
|
|
192
|
+
if not isinstance(data, dict) or not isinstance(data.get("cases"), list):
|
|
193
|
+
raise EvalConfigError(f"results file {results_path} has no 'cases' list")
|
|
194
|
+
|
|
195
|
+
clusters = cluster_results(data)
|
|
196
|
+
flagged = sorted({cid for c in clusters for cid in c["case_ids"]})
|
|
197
|
+
analysis: dict[str, Any] = {
|
|
198
|
+
"results_file": str(results_path),
|
|
199
|
+
"dataset_hash": data.get("dataset_hash"),
|
|
200
|
+
"generated_at": utc_now_iso(),
|
|
201
|
+
"total_cases": len(data["cases"]),
|
|
202
|
+
"flagged_cases": len(flagged),
|
|
203
|
+
"summary": data.get("summary"),
|
|
204
|
+
"clusters": clusters,
|
|
205
|
+
"judge_summary": None,
|
|
206
|
+
}
|
|
207
|
+
|
|
208
|
+
if judge and clusters:
|
|
209
|
+
if project_root is None:
|
|
210
|
+
raise EvalConfigError("--judge needs the project's environment; run inside the project")
|
|
211
|
+
config = load_eval_config(_paths.default_eval_config(project_root))
|
|
212
|
+
identity = resolve_judge_identity(
|
|
213
|
+
project_root,
|
|
214
|
+
config_provider=config.judge_provider,
|
|
215
|
+
config_model=config.judge_model,
|
|
216
|
+
flag_provider=judge_provider,
|
|
217
|
+
flag_model=judge_model,
|
|
218
|
+
)
|
|
219
|
+
console.print("Asking the judge model to summarise the clusters...")
|
|
220
|
+
output = run_judge_runner(
|
|
221
|
+
project_root,
|
|
222
|
+
{
|
|
223
|
+
"judge": identity,
|
|
224
|
+
"items": [
|
|
225
|
+
{
|
|
226
|
+
"id": "summary",
|
|
227
|
+
"kind": "summarize",
|
|
228
|
+
"prompt": _summary_prompt(data, clusters),
|
|
229
|
+
}
|
|
230
|
+
],
|
|
231
|
+
},
|
|
232
|
+
)
|
|
233
|
+
entry = results_by_id(output).get("summary") or {}
|
|
234
|
+
if entry.get("error"):
|
|
235
|
+
raise EvalConfigError(f"judge summary failed: {entry['error']}")
|
|
236
|
+
analysis["judge_summary"] = entry.get("reasoning") or ""
|
|
237
|
+
analysis["judge"] = {"provider": output.get("provider"), "model": output.get("model")}
|
|
238
|
+
|
|
239
|
+
root = project_root or Path.cwd()
|
|
240
|
+
out = Path(output_path) if output_path else _paths.default_analysis_path(root)
|
|
241
|
+
if not out.is_absolute():
|
|
242
|
+
out = root / out
|
|
243
|
+
write_json_file(out, analysis)
|
|
244
|
+
|
|
245
|
+
shown = clusters[:top_k] if top_k else clusters
|
|
246
|
+
if not clusters:
|
|
247
|
+
console.print(
|
|
248
|
+
f"[green]No failures to analyze[/green] in {results_path} ({len(data['cases'])} case(s) passed)."
|
|
249
|
+
)
|
|
250
|
+
else:
|
|
251
|
+
table = Table(
|
|
252
|
+
title=f"Failure clusters ({len(flagged)} of {len(data['cases'])} cases)",
|
|
253
|
+
show_header=True,
|
|
254
|
+
header_style="bold",
|
|
255
|
+
)
|
|
256
|
+
table.add_column("#", justify="right")
|
|
257
|
+
table.add_column("Status")
|
|
258
|
+
table.add_column("Check / metric")
|
|
259
|
+
table.add_column("Pattern")
|
|
260
|
+
table.add_column("Cases", justify="right")
|
|
261
|
+
table.add_column("Case ids")
|
|
262
|
+
for cluster in shown:
|
|
263
|
+
ids = ", ".join(cluster["case_ids"][:5]) + (
|
|
264
|
+
", ..." if len(cluster["case_ids"]) > 5 else ""
|
|
265
|
+
)
|
|
266
|
+
table.add_row(
|
|
267
|
+
str(cluster["rank"]),
|
|
268
|
+
cluster["status"],
|
|
269
|
+
cluster["kind"],
|
|
270
|
+
cluster["pattern"],
|
|
271
|
+
str(cluster["count"]),
|
|
272
|
+
ids,
|
|
273
|
+
)
|
|
274
|
+
console.print(table)
|
|
275
|
+
if analysis["judge_summary"]:
|
|
276
|
+
console.rule("[bold]Judge summary[/bold]")
|
|
277
|
+
console.print(analysis["judge_summary"])
|
|
278
|
+
console.print(f"Analysis saved to [green]{out}[/green]")
|
|
@@ -0,0 +1,284 @@
|
|
|
1
|
+
# Copyright 2026 Google LLC
|
|
2
|
+
# Modifications Copyright 2026 graph-agents-cli contributors
|
|
3
|
+
#
|
|
4
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
5
|
+
# you may not use this file except in compliance with the License.
|
|
6
|
+
# You may obtain a copy of the License at
|
|
7
|
+
#
|
|
8
|
+
# https://www.apache.org/licenses/LICENSE-2.0
|
|
9
|
+
#
|
|
10
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
11
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
12
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
13
|
+
# See the License for the specific language governing permissions and
|
|
14
|
+
# limitations under the License.
|
|
15
|
+
|
|
16
|
+
"""graph-agents-cli eval compare command: diff two results files."""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import json
|
|
21
|
+
import sys
|
|
22
|
+
from pathlib import Path
|
|
23
|
+
from typing import Any
|
|
24
|
+
|
|
25
|
+
import click
|
|
26
|
+
from rich.table import Table
|
|
27
|
+
|
|
28
|
+
from graph_agents_cli._output import Console, emit
|
|
29
|
+
from graph_agents_cli.eval._common import EXIT_GATE_FAILED, load_json_file
|
|
30
|
+
from graph_agents_cli.eval.gate import STATUS_RANK
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def _diff(base: dict, cand: dict, prefix: str = "") -> dict:
|
|
34
|
+
"""Compute a recursive diff between two dicts.
|
|
35
|
+
|
|
36
|
+
Nested dicts are diffed recursively with dotted key paths.
|
|
37
|
+
Numeric changes include a delta (e.g., "+0.07" or "-0.03").
|
|
38
|
+
"""
|
|
39
|
+
differences = {}
|
|
40
|
+
|
|
41
|
+
all_keys = set(base.keys()) | set(cand.keys())
|
|
42
|
+
for key in sorted(all_keys):
|
|
43
|
+
full_key = f"{prefix}{key}" if not prefix else f"{prefix}.{key}"
|
|
44
|
+
base_val = base.get(key)
|
|
45
|
+
cand_val = cand.get(key)
|
|
46
|
+
|
|
47
|
+
if base_val == cand_val:
|
|
48
|
+
continue
|
|
49
|
+
|
|
50
|
+
# Recurse into nested dicts
|
|
51
|
+
if isinstance(base_val, dict) and isinstance(cand_val, dict):
|
|
52
|
+
nested = _diff(base_val, cand_val, prefix=full_key)
|
|
53
|
+
differences.update(nested["differences"])
|
|
54
|
+
continue
|
|
55
|
+
|
|
56
|
+
entry = {"baseline": base_val, "candidate": cand_val}
|
|
57
|
+
if isinstance(base_val, int | float) and isinstance(cand_val, int | float):
|
|
58
|
+
delta = cand_val - base_val
|
|
59
|
+
entry["delta"] = f"+{delta}" if delta >= 0 else str(delta)
|
|
60
|
+
differences[full_key] = entry
|
|
61
|
+
|
|
62
|
+
if prefix:
|
|
63
|
+
return {"differences": differences}
|
|
64
|
+
|
|
65
|
+
return {
|
|
66
|
+
"baseline_keys": sorted(base.keys()),
|
|
67
|
+
"candidate_keys": sorted(cand.keys()),
|
|
68
|
+
"differences": differences,
|
|
69
|
+
"changed_keys": sorted(differences.keys()),
|
|
70
|
+
"unchanged_keys": sorted(k for k in all_keys if k not in differences),
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def _rank(status: str | None) -> int:
|
|
75
|
+
return STATUS_RANK.get(status or "", 3)
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def _score_map(case: dict[str, Any]) -> dict[str, float | None]:
|
|
79
|
+
return {
|
|
80
|
+
name: entry.get("score")
|
|
81
|
+
for name, entry in (case.get("judge_scores") or {}).items()
|
|
82
|
+
if isinstance(entry, dict)
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def compare_results(base: dict[str, Any], cand: dict[str, Any]) -> dict[str, Any]:
|
|
87
|
+
"""Per-case status changes, quality-rate deltas, summary deltas, regression flag."""
|
|
88
|
+
base_cases = {c["id"]: c for c in base.get("cases", []) if isinstance(c, dict)}
|
|
89
|
+
cand_cases = {c["id"]: c for c in cand.get("cases", []) if isinstance(c, dict)}
|
|
90
|
+
cases: dict[str, dict[str, Any]] = {}
|
|
91
|
+
regressions: list[str] = []
|
|
92
|
+
improvements: list[str] = []
|
|
93
|
+
for case_id in sorted(set(base_cases) | set(cand_cases)):
|
|
94
|
+
b = base_cases.get(case_id)
|
|
95
|
+
c = cand_cases.get(case_id)
|
|
96
|
+
b_status = b.get("status") if b else None
|
|
97
|
+
c_status = c.get("status") if c else None
|
|
98
|
+
if b is None:
|
|
99
|
+
change = "added"
|
|
100
|
+
elif c is None:
|
|
101
|
+
change = "removed"
|
|
102
|
+
regressions.append(f"{case_id}: removed from the candidate run")
|
|
103
|
+
elif _rank(c_status) > _rank(b_status):
|
|
104
|
+
change = "regressed"
|
|
105
|
+
regressions.append(f"{case_id}: {b_status} -> {c_status}")
|
|
106
|
+
elif _rank(c_status) < _rank(b_status):
|
|
107
|
+
change = "improved"
|
|
108
|
+
improvements.append(f"{case_id}: {b_status} -> {c_status}")
|
|
109
|
+
else:
|
|
110
|
+
change = "unchanged"
|
|
111
|
+
scores: dict[str, dict[str, Any]] = {}
|
|
112
|
+
b_scores = _score_map(b) if b else {}
|
|
113
|
+
c_scores = _score_map(c) if c else {}
|
|
114
|
+
for metric in sorted(set(b_scores) | set(c_scores)):
|
|
115
|
+
entry: dict[str, Any] = {
|
|
116
|
+
"baseline": b_scores.get(metric),
|
|
117
|
+
"candidate": c_scores.get(metric),
|
|
118
|
+
}
|
|
119
|
+
if isinstance(entry["baseline"], int | float) and isinstance(
|
|
120
|
+
entry["candidate"], int | float
|
|
121
|
+
):
|
|
122
|
+
entry["delta"] = round(entry["candidate"] - entry["baseline"], 4)
|
|
123
|
+
scores[metric] = entry
|
|
124
|
+
cases[case_id] = {
|
|
125
|
+
"baseline": b_status,
|
|
126
|
+
"candidate": c_status,
|
|
127
|
+
"change": change,
|
|
128
|
+
"candidate_reasons": (c or {}).get("reasons", []),
|
|
129
|
+
"scores": scores,
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
quality: dict[str, dict[str, Any]] = {}
|
|
133
|
+
b_quality = base.get("quality") or {}
|
|
134
|
+
c_quality = cand.get("quality") or {}
|
|
135
|
+
for metric in sorted(set(b_quality) | set(c_quality)):
|
|
136
|
+
b_entry = b_quality.get(metric) or {}
|
|
137
|
+
c_entry = c_quality.get(metric) or {}
|
|
138
|
+
b_rate = b_entry.get("pass_rate")
|
|
139
|
+
c_rate = c_entry.get("pass_rate")
|
|
140
|
+
entry = {
|
|
141
|
+
"baseline_pass_rate": b_rate,
|
|
142
|
+
"candidate_pass_rate": c_rate,
|
|
143
|
+
"baseline_met": b_entry.get("met"),
|
|
144
|
+
"candidate_met": c_entry.get("met"),
|
|
145
|
+
"min_pass_rate": c_entry.get("min_pass_rate", b_entry.get("min_pass_rate")),
|
|
146
|
+
}
|
|
147
|
+
if isinstance(b_rate, int | float) and isinstance(c_rate, int | float):
|
|
148
|
+
entry["delta"] = round(c_rate - b_rate, 4)
|
|
149
|
+
if c_rate < b_rate:
|
|
150
|
+
regressions.append(f"quality {metric}: pass rate {b_rate:.0%} -> {c_rate:.0%}")
|
|
151
|
+
elif c_rate > b_rate:
|
|
152
|
+
improvements.append(f"quality {metric}: pass rate {b_rate:.0%} -> {c_rate:.0%}")
|
|
153
|
+
if b_entry.get("met") is True and c_entry.get("met") is False:
|
|
154
|
+
regressions.append(f"quality {metric}: min_pass_rate no longer met")
|
|
155
|
+
quality[metric] = entry
|
|
156
|
+
|
|
157
|
+
b_summary = base.get("summary") or {}
|
|
158
|
+
c_summary = cand.get("summary") or {}
|
|
159
|
+
b_exit = b_summary.get("exit_code")
|
|
160
|
+
c_exit = c_summary.get("exit_code")
|
|
161
|
+
if isinstance(b_exit, int) and isinstance(c_exit, int) and c_exit > b_exit:
|
|
162
|
+
regressions.append(f"exit code {b_exit} -> {c_exit}")
|
|
163
|
+
|
|
164
|
+
return {
|
|
165
|
+
"dataset_hash": {
|
|
166
|
+
"baseline": base.get("dataset_hash"),
|
|
167
|
+
"candidate": cand.get("dataset_hash"),
|
|
168
|
+
},
|
|
169
|
+
"same_dataset": base.get("dataset_hash") == cand.get("dataset_hash"),
|
|
170
|
+
"traces_dataset_hash": {
|
|
171
|
+
label: {
|
|
172
|
+
"value": doc.get("traces_dataset_hash"),
|
|
173
|
+
# Only a recorded, differing value is a mismatch (older results lack the key).
|
|
174
|
+
"mismatch": bool(
|
|
175
|
+
doc.get("traces_dataset_hash")
|
|
176
|
+
and doc.get("dataset_hash")
|
|
177
|
+
and doc.get("traces_dataset_hash") != doc.get("dataset_hash")
|
|
178
|
+
),
|
|
179
|
+
}
|
|
180
|
+
for label, doc in (("baseline", base), ("candidate", cand))
|
|
181
|
+
},
|
|
182
|
+
"summary": _diff(b_summary, c_summary),
|
|
183
|
+
"quality": quality,
|
|
184
|
+
"cases": cases,
|
|
185
|
+
"regressions": regressions,
|
|
186
|
+
"improvements": improvements,
|
|
187
|
+
"regressed": bool(regressions),
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
def _print_report(console: Console, report: dict[str, Any], baseline: str, candidate: str) -> None:
|
|
192
|
+
console.print(f"Baseline: [cyan]{baseline}[/cyan]")
|
|
193
|
+
console.print(f"Candidate: [cyan]{candidate}[/cyan]")
|
|
194
|
+
if not report["same_dataset"]:
|
|
195
|
+
console.print("[yellow]Warning:[/yellow] the two results come from different datasets.")
|
|
196
|
+
for label, entry in (
|
|
197
|
+
("baseline", report["traces_dataset_hash"]["baseline"]),
|
|
198
|
+
("candidate", report["traces_dataset_hash"]["candidate"]),
|
|
199
|
+
):
|
|
200
|
+
if entry["mismatch"]:
|
|
201
|
+
console.print(
|
|
202
|
+
f"[yellow]Warning:[/yellow] the {label} results were graded against a dataset "
|
|
203
|
+
"the traces were not generated from (traces_dataset_hash != dataset_hash)."
|
|
204
|
+
)
|
|
205
|
+
|
|
206
|
+
changed = {k: v for k, v in report["cases"].items() if v["change"] != "unchanged"}
|
|
207
|
+
table = Table(title="Case status changes", show_header=True, header_style="bold")
|
|
208
|
+
table.add_column("Case")
|
|
209
|
+
table.add_column("Baseline")
|
|
210
|
+
table.add_column("Candidate")
|
|
211
|
+
table.add_column("Change")
|
|
212
|
+
style = {"regressed": "red", "improved": "green", "added": "cyan", "removed": "magenta"}
|
|
213
|
+
for case_id, entry in changed.items():
|
|
214
|
+
table.add_row(
|
|
215
|
+
case_id,
|
|
216
|
+
str(entry["baseline"]),
|
|
217
|
+
str(entry["candidate"]),
|
|
218
|
+
f"[{style[entry['change']]}]{entry['change']}[/{style[entry['change']]}]",
|
|
219
|
+
)
|
|
220
|
+
if changed:
|
|
221
|
+
console.print(table)
|
|
222
|
+
else:
|
|
223
|
+
console.print(f"No case status changes across {len(report['cases'])} case(s).")
|
|
224
|
+
|
|
225
|
+
if report["quality"]:
|
|
226
|
+
qtable = Table(title="Quality metrics", show_header=True, header_style="bold")
|
|
227
|
+
qtable.add_column("Metric")
|
|
228
|
+
qtable.add_column("Baseline", justify="right")
|
|
229
|
+
qtable.add_column("Candidate", justify="right")
|
|
230
|
+
qtable.add_column("Delta", justify="right")
|
|
231
|
+
for metric, entry in report["quality"].items():
|
|
232
|
+
b, c = entry["baseline_pass_rate"], entry["candidate_pass_rate"]
|
|
233
|
+
qtable.add_row(
|
|
234
|
+
metric,
|
|
235
|
+
"n/a" if b is None else f"{b:.0%}",
|
|
236
|
+
"n/a" if c is None else f"{c:.0%}",
|
|
237
|
+
"n/a" if "delta" not in entry else f"{entry['delta']:+.0%}",
|
|
238
|
+
)
|
|
239
|
+
console.print(qtable)
|
|
240
|
+
|
|
241
|
+
diffs = report["summary"]["differences"]
|
|
242
|
+
if diffs:
|
|
243
|
+
console.print("Summary deltas:")
|
|
244
|
+
for key, entry in diffs.items():
|
|
245
|
+
delta = f" ({entry['delta']})" if "delta" in entry else ""
|
|
246
|
+
console.print(f" {key}: {entry['baseline']} -> {entry['candidate']}{delta}")
|
|
247
|
+
if report["regressions"]:
|
|
248
|
+
console.print("[red]Regressions:[/red]")
|
|
249
|
+
for line in report["regressions"]:
|
|
250
|
+
console.print(f" - {line}")
|
|
251
|
+
if report["improvements"]:
|
|
252
|
+
console.print("[green]Improvements:[/green]")
|
|
253
|
+
for line in report["improvements"]:
|
|
254
|
+
console.print(f" - {line}")
|
|
255
|
+
if not report["regressions"]:
|
|
256
|
+
console.print("[green]No regressions.[/green]")
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
@click.command("compare")
|
|
260
|
+
@click.argument("baseline", type=click.Path(exists=True, dir_okay=False))
|
|
261
|
+
@click.argument("candidate", type=click.Path(exists=True, dir_okay=False))
|
|
262
|
+
@click.option("--fail-on-regression", is_flag=True, help="Exit 1 when the candidate regresses.")
|
|
263
|
+
@click.option(
|
|
264
|
+
"--json", "as_json", is_flag=True, help="Print the full comparison as JSON instead of tables."
|
|
265
|
+
)
|
|
266
|
+
def cmd_compare(baseline: str, candidate: str, *, fail_on_regression: bool, as_json: bool) -> None:
|
|
267
|
+
"""Compare two eval results files (baseline, candidate).
|
|
268
|
+
|
|
269
|
+
Reports per-case status changes, quality pass-rate deltas and summary
|
|
270
|
+
deltas. A regression is a case whose status got worse (or disappeared), a
|
|
271
|
+
quality metric whose pass rate dropped or stopped meeting its
|
|
272
|
+
min_pass_rate, or a worse exit code. Purely in-process.
|
|
273
|
+
"""
|
|
274
|
+
base = load_json_file(Path(baseline), "baseline results")
|
|
275
|
+
cand = load_json_file(Path(candidate), "candidate results")
|
|
276
|
+
report = compare_results(base, cand)
|
|
277
|
+
report["baseline"] = baseline
|
|
278
|
+
report["candidate"] = candidate
|
|
279
|
+
if as_json:
|
|
280
|
+
emit(json.loads(json.dumps(report)))
|
|
281
|
+
else:
|
|
282
|
+
_print_report(Console(), report, baseline, candidate)
|
|
283
|
+
if fail_on_regression and report["regressed"]:
|
|
284
|
+
sys.exit(EXIT_GATE_FAILED)
|
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
# Copyright 2026 Google LLC
|
|
2
|
+
# Modifications Copyright 2026 graph-agents-cli contributors
|
|
3
|
+
#
|
|
4
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
5
|
+
# you may not use this file except in compliance with the License.
|
|
6
|
+
# You may obtain a copy of the License at
|
|
7
|
+
#
|
|
8
|
+
# https://www.apache.org/licenses/LICENSE-2.0
|
|
9
|
+
#
|
|
10
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
11
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
12
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
13
|
+
# See the License for the specific language governing permissions and
|
|
14
|
+
# limitations under the License.
|
|
15
|
+
|
|
16
|
+
"""graph-agents-cli eval command group."""
|
|
17
|
+
|
|
18
|
+
import click
|
|
19
|
+
|
|
20
|
+
from graph_agents_cli._click import LazyGroup
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
@click.group("eval", cls=LazyGroup)
|
|
24
|
+
def eval_group():
|
|
25
|
+
"""Evaluate agents and compare results.
|
|
26
|
+
|
|
27
|
+
\b
|
|
28
|
+
Core:
|
|
29
|
+
run Chain generate + grade in one command
|
|
30
|
+
generate Run the agent over the eval dataset and write traces
|
|
31
|
+
grade Grade traces against checks and judges; apply the gate
|
|
32
|
+
compare Compare two eval results files
|
|
33
|
+
metric Discover evaluation metrics
|
|
34
|
+
|
|
35
|
+
\b
|
|
36
|
+
Analysis and sharing:
|
|
37
|
+
analyze Cluster failures by reason (optionally judge-summarised)
|
|
38
|
+
submit Upload a dataset and results to LangSmith (optional)
|
|
39
|
+
|
|
40
|
+
\b
|
|
41
|
+
Exit codes (run, generate, grade):
|
|
42
|
+
0 gate met, 1 gate failed, 2 incomplete run, 3 configuration error
|
|
43
|
+
"""
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
eval_group.add_lazy_command(
|
|
47
|
+
"run",
|
|
48
|
+
"graph_agents_cli.eval.cmd_run:cmd_run",
|
|
49
|
+
"Chain `eval generate` and `eval grade` in one command.",
|
|
50
|
+
)
|
|
51
|
+
eval_group.add_lazy_command(
|
|
52
|
+
"generate",
|
|
53
|
+
"graph_agents_cli.eval.cmd_generate:cmd_generate",
|
|
54
|
+
"Run the agent over the eval dataset and write traces.",
|
|
55
|
+
)
|
|
56
|
+
eval_group.add_lazy_command(
|
|
57
|
+
"grade",
|
|
58
|
+
"graph_agents_cli.eval.cmd_grade:cmd_grade",
|
|
59
|
+
"Grade traces against the dataset's checks and judges and apply the gate.",
|
|
60
|
+
)
|
|
61
|
+
eval_group.add_lazy_command(
|
|
62
|
+
"compare",
|
|
63
|
+
"graph_agents_cli.eval.cmd_compare:cmd_compare",
|
|
64
|
+
"Compare two eval results files (baseline, candidate).",
|
|
65
|
+
)
|
|
66
|
+
eval_group.add_lazy_command(
|
|
67
|
+
"analyze",
|
|
68
|
+
"graph_agents_cli.eval.cmd_analyze:cmd_analyze",
|
|
69
|
+
"Cluster failed, quality-below-threshold, error and missing cases by reason.",
|
|
70
|
+
)
|
|
71
|
+
eval_group.add_lazy_command(
|
|
72
|
+
"submit",
|
|
73
|
+
"graph_agents_cli.eval.cmd_submit:cmd_submit",
|
|
74
|
+
"Upload the dataset and a results file to LangSmith as an experiment.",
|
|
75
|
+
)
|
|
76
|
+
eval_group.add_lazy_command(
|
|
77
|
+
"metric",
|
|
78
|
+
"graph_agents_cli.eval.cmd_metric:metric_group",
|
|
79
|
+
"Discover evaluation metrics: deterministic checks and judges.",
|
|
80
|
+
)
|