graph-agents-cli 0.3.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- graph_agents_cli/__init__.py +26 -0
- graph_agents_cli/_api_policy.py +2145 -0
- graph_agents_cli/_approvals.py +400 -0
- graph_agents_cli/_build.py +186 -0
- graph_agents_cli/_build_info.json +7 -0
- graph_agents_cli/_chat_client.py +462 -0
- graph_agents_cli/_click.py +157 -0
- graph_agents_cli/_defaults.py +139 -0
- graph_agents_cli/_experiments.py +64 -0
- graph_agents_cli/_http.py +192 -0
- graph_agents_cli/_output.py +83 -0
- graph_agents_cli/_project.py +462 -0
- graph_agents_cli/_remote.py +220 -0
- graph_agents_cli/_response_schema.py +264 -0
- graph_agents_cli/_runner.py +319 -0
- graph_agents_cli/_skills_check.py +274 -0
- graph_agents_cli/_tools.py +189 -0
- graph_agents_cli/_trust.py +66 -0
- graph_agents_cli/api/__init__.py +15 -0
- graph_agents_cli/api/_changes.py +506 -0
- graph_agents_cli/api/_files.py +658 -0
- graph_agents_cli/api/cmd_api.py +2480 -0
- graph_agents_cli/deploy/__init__.py +15 -0
- graph_agents_cli/deploy/_config.py +171 -0
- graph_agents_cli/deploy/_image.py +128 -0
- graph_agents_cli/deploy/_kube.py +286 -0
- graph_agents_cli/deploy/_modes.py +234 -0
- graph_agents_cli/deploy/_preflight.py +370 -0
- graph_agents_cli/deploy/_values.py +168 -0
- graph_agents_cli/deploy/cmd_deploy.py +1866 -0
- graph_agents_cli/deploy/gitops.py +562 -0
- graph_agents_cli/deploy/local_load.py +273 -0
- graph_agents_cli/dev/__init__.py +13 -0
- graph_agents_cli/dev/cmd_build.py +131 -0
- graph_agents_cli/dev/cmd_install.py +78 -0
- graph_agents_cli/dev/cmd_lint.py +119 -0
- graph_agents_cli/dev/cmd_playground.py +297 -0
- graph_agents_cli/dev/policy_check.py +1287 -0
- graph_agents_cli/eval/__init__.py +22 -0
- graph_agents_cli/eval/_client.py +670 -0
- graph_agents_cli/eval/_common.py +177 -0
- graph_agents_cli/eval/_judge.py +168 -0
- graph_agents_cli/eval/_judge_runner.py +238 -0
- graph_agents_cli/eval/_paths.py +212 -0
- graph_agents_cli/eval/checks.py +581 -0
- graph_agents_cli/eval/cmd_analyze.py +278 -0
- graph_agents_cli/eval/cmd_compare.py +284 -0
- graph_agents_cli/eval/cmd_eval_group.py +80 -0
- graph_agents_cli/eval/cmd_generate.py +558 -0
- graph_agents_cli/eval/cmd_grade.py +466 -0
- graph_agents_cli/eval/cmd_metric.py +156 -0
- graph_agents_cli/eval/cmd_run.py +370 -0
- graph_agents_cli/eval/cmd_submit.py +400 -0
- graph_agents_cli/eval/config.py +435 -0
- graph_agents_cli/eval/dataset.py +350 -0
- graph_agents_cli/eval/gate.py +420 -0
- graph_agents_cli/eval/transcript.py +192 -0
- graph_agents_cli/extension/__init__.py +13 -0
- graph_agents_cli/extension/_compat.py +86 -0
- graph_agents_cli/extension/_loader.py +293 -0
- graph_agents_cli/extension/_manifest.py +135 -0
- graph_agents_cli/extension/_overrides.py +195 -0
- graph_agents_cli/extension/_paths.py +91 -0
- graph_agents_cli/extension/_refs.py +193 -0
- graph_agents_cli/extension/_resolver.py +453 -0
- graph_agents_cli/extension/_schema.py +106 -0
- graph_agents_cli/extension/_spec.py +253 -0
- graph_agents_cli/extension/_sync.py +102 -0
- graph_agents_cli/extension/_trust.py +58 -0
- graph_agents_cli/extension/cmd_extension_add.py +259 -0
- graph_agents_cli/extension/cmd_extension_group.py +57 -0
- graph_agents_cli/extension/cmd_extension_list.py +56 -0
- graph_agents_cli/extension/cmd_extension_remove.py +61 -0
- graph_agents_cli/extension/cmd_extension_update.py +195 -0
- graph_agents_cli/info/__init__.py +13 -0
- graph_agents_cli/info/cmd_info.py +222 -0
- graph_agents_cli/infra/__init__.py +15 -0
- graph_agents_cli/infra/checks.py +1169 -0
- graph_agents_cli/infra/cmd_infra.py +103 -0
- graph_agents_cli/main.py +591 -0
- graph_agents_cli/peer/__init__.py +15 -0
- graph_agents_cli/peer/_generate.py +254 -0
- graph_agents_cli/peer/cmd_peer.py +1151 -0
- graph_agents_cli/run/__init__.py +13 -0
- graph_agents_cli/run/_local_server.py +1157 -0
- graph_agents_cli/run/_signals.py +141 -0
- graph_agents_cli/run/cmd_approvals.py +530 -0
- graph_agents_cli/run/cmd_run.py +1421 -0
- graph_agents_cli/scaffold/__init__.py +19 -0
- graph_agents_cli/scaffold/agents/README.md +24 -0
- graph_agents_cli/scaffold/agents/empty_py/.template/templateconfig.yaml +22 -0
- graph_agents_cli/scaffold/agents/langgraph/.env.example +292 -0
- graph_agents_cli/scaffold/agents/langgraph/.template/templateconfig.yaml +28 -0
- graph_agents_cli/scaffold/agents/langgraph/Dockerfile +59 -0
- graph_agents_cli/scaffold/agents/langgraph/Dockerfile.langgraph-server +59 -0
- graph_agents_cli/scaffold/agents/langgraph/README.md +571 -0
- graph_agents_cli/scaffold/agents/langgraph/api-policy.yaml +60 -0
- graph_agents_cli/scaffold/agents/langgraph/app/__init__.py +20 -0
- graph_agents_cli/scaffold/agents/langgraph/app/agent.py +174 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/__init__.py +15 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/a2a.py +2162 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/a2a_client.py +1167 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/api_client.py +4220 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/approvals.py +1349 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/auth.py +1986 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/chat.py +2962 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/checkpointer.py +432 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/content.py +569 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/db.py +580 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/limits.py +203 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/metrics.py +231 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/middleware.py +361 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/model.py +611 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/playground.py +230 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/run_locks.py +459 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/structured.py +755 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/telemetry.py +681 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/threads.py +493 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/token_exchange.py +959 -0
- graph_agents_cli/scaffold/agents/langgraph/app/fast_api_app.py +770 -0
- graph_agents_cli/scaffold/agents/langgraph/app/policies/__init__.py +55 -0
- graph_agents_cli/scaffold/agents/langgraph/app/policies/custom.py +97 -0
- graph_agents_cli/scaffold/agents/langgraph/app/tools/__init__.py +46 -0
- graph_agents_cli/scaffold/agents/langgraph/app/tools/example_api.py +92 -0
- graph_agents_cli/scaffold/agents/langgraph/app/tools/weather.py +33 -0
- graph_agents_cli/scaffold/agents/langgraph/langgraph.json +14 -0
- graph_agents_cli/scaffold/agents/langgraph/pyproject.toml +78 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/conftest.py +376 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/eval/datasets/basic-dataset.json +53 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/eval/eval_config.yaml +32 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/approval_graph.py +137 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/fake_issuer.py +216 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/fake_openai.py +357 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_a2a_outcomes.py +569 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_a2a_relay.py +479 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_api_surface.py +812 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_approvals.py +1367 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_approvals_server.py +794 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_cross_actor.py +497 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_cross_actor_server.py +247 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_history_repair.py +278 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_model_apis.py +242 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_postgres.py +637 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_resilience_postgres.py +770 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_runtime_guardrails.py +854 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_server_e2e.py +340 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_server_runtime.py +989 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_structured_answers.py +584 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_structured_server.py +222 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_token_exchange_issuer.py +650 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/load_test/.results/.placeholder +0 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/load_test/README.md +22 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/load_test/conftest.py +21 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/load_test/load_test.py +81 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_a2a_client.py +824 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_a2a_scoping.py +724 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_api_client.py +1214 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_api_client_hardening.py +716 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_api_policy_rpc.py +767 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_approval_ledger.py +1536 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_fake_model.py +115 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_jwt_policy.py +991 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_limits.py +310 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_logging.py +148 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_logging_hardening.py +271 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_policy.py +378 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_resilience.py +610 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_server_auth.py +702 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_structured.py +673 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_telemetry.py +404 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_thread_listing.py +255 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_threads.py +268 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_token_exchange.py +1320 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_untrusted_content.py +393 -0
- graph_agents_cli/scaffold/agents/langgraph/uv-fastapi.lock +2084 -0
- graph_agents_cli/scaffold/agents/langgraph/uv-langgraph-server.lock +2106 -0
- graph_agents_cli/scaffold/agents/langgraph/{{cookiecutter.agent_guidance_filename}} +129 -0
- graph_agents_cli/scaffold/base_templates/_shared/graph-agents-cli-manifest.yaml +36 -0
- graph_agents_cli/scaffold/base_templates/python/.dockerignore +32 -0
- graph_agents_cli/scaffold/base_templates/python/.github/CODEOWNERS +30 -0
- graph_agents_cli/scaffold/base_templates/python/.github/agent.env +7 -0
- graph_agents_cli/scaffold/base_templates/python/.github/workflows/pr_checks.yaml +214 -0
- graph_agents_cli/scaffold/base_templates/python/.gitignore +209 -0
- graph_agents_cli/scaffold/base_templates/python/tests/unit/test_dummy.py +23 -0
- graph_agents_cli/scaffold/base_templates/python/{{cookiecutter.agent_guidance_filename}} +35 -0
- graph_agents_cli/scaffold/cmd_scaffold_group.py +49 -0
- graph_agents_cli/scaffold/commands/__init__.py +13 -0
- graph_agents_cli/scaffold/commands/create.py +1424 -0
- graph_agents_cli/scaffold/commands/enhance.py +1652 -0
- graph_agents_cli/scaffold/commands/upgrade.py +570 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/.github/agent.env +12 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/.github/workflows/promote-to-prod.yaml +371 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/.github/workflows/staging.yaml +450 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/argocd/application-dev.yaml +43 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/argocd/application-prod.yaml +41 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/argocd/application-staging.yaml +43 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/.helmignore +14 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/Chart.yaml +21 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/examples/networkpolicy.yaml +103 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/NOTES.txt +48 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/_helpers.tpl +189 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/certificate.yaml +15 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/configmap.yaml +10 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/deployment.yaml +199 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/hpa.yaml +22 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/httproute.yaml +30 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/ingress.yaml +39 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/networkpolicy.yaml +48 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/pdb.yaml +13 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/postgresql-secret.yaml +37 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/service.yaml +15 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/serviceaccount.yaml +13 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/servicemonitor.yaml +42 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values-dev.yaml +22 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values-prod.yaml +45 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values-staging.yaml +29 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values.yaml +396 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/tests/integration/test_chart.py +269 -0
- graph_agents_cli/scaffold/deployment_targets/none/README.md +5 -0
- graph_agents_cli/scaffold/deployment_targets/none/python/README.md +6 -0
- graph_agents_cli/scaffold/utils/__init__.py +13 -0
- graph_agents_cli/scaffold/utils/backup.py +212 -0
- graph_agents_cli/scaffold/utils/build_record.py +257 -0
- graph_agents_cli/scaffold/utils/cli_options.py +184 -0
- graph_agents_cli/scaffold/utils/fs.py +83 -0
- graph_agents_cli/scaffold/utils/generate_locks.py +214 -0
- graph_agents_cli/scaffold/utils/generation_metadata.py +88 -0
- graph_agents_cli/scaffold/utils/keyedit.py +768 -0
- graph_agents_cli/scaffold/utils/keymerge.py +537 -0
- graph_agents_cli/scaffold/utils/language.py +138 -0
- graph_agents_cli/scaffold/utils/lock_utils.py +94 -0
- graph_agents_cli/scaffold/utils/logging.py +77 -0
- graph_agents_cli/scaffold/utils/manifest.py +292 -0
- graph_agents_cli/scaffold/utils/merge.py +970 -0
- graph_agents_cli/scaffold/utils/merge3.py +216 -0
- graph_agents_cli/scaffold/utils/openapi_seed.py +199 -0
- graph_agents_cli/scaffold/utils/remote_template.py +376 -0
- graph_agents_cli/scaffold/utils/template.py +1352 -0
- graph_agents_cli/scaffold/utils/upgrade.py +894 -0
- graph_agents_cli/scaffold/utils/version.py +438 -0
- graph_agents_cli/secrets/__init__.py +15 -0
- graph_agents_cli/secrets/_apply.py +954 -0
- graph_agents_cli/secrets/_required.py +188 -0
- graph_agents_cli/secrets/cmd_secrets.py +211 -0
- graph_agents_cli/setup/__init__.py +13 -0
- graph_agents_cli/setup/_antigravity.py +221 -0
- graph_agents_cli/setup/cmd_auth.py +1030 -0
- graph_agents_cli/setup/cmd_dev_token.py +513 -0
- graph_agents_cli/setup/cmd_setup.py +428 -0
- graph_agents_cli/setup/cmd_update.py +140 -0
- graph_agents_cli/skills/__init__.py +13 -0
- graph_agents_cli/skills/_bundle.py +65 -0
- graph_agents_cli/skills/data/README.md +19 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/SKILL.md +357 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/references/github-settings.md +113 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/references/gitops.md +137 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/references/kubernetes.md +315 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/references/secrets.md +160 -0
- graph_agents_cli/skills/data/graph-agents-cli-eval/SKILL.md +303 -0
- graph_agents_cli/skills/data/graph-agents-cli-eval/references/dataset_schema.md +282 -0
- graph_agents_cli/skills/data/graph-agents-cli-eval/references/metrics-guide.md +143 -0
- graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/SKILL.md +659 -0
- graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/references/langchain-models.md +124 -0
- graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/references/langgraph.md +235 -0
- graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/references/template-contract.md +477 -0
- graph_agents_cli/skills/data/graph-agents-cli-observability/SKILL.md +231 -0
- graph_agents_cli/skills/data/graph-agents-cli-observability/references/langsmith.md +46 -0
- graph_agents_cli/skills/data/graph-agents-cli-observability/references/otel.md +59 -0
- graph_agents_cli/skills/data/graph-agents-cli-scaffold/SKILL.md +414 -0
- graph_agents_cli/skills/data/graph-agents-cli-scaffold/references/flags.md +134 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/SKILL.md +478 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/brainstorming.md +118 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/commands.md +419 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/extension.md +156 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/internals.md +67 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/spec-template.md +56 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/terminology.md +119 -0
- graph_agents_cli/system/__init__.py +15 -0
- graph_agents_cli/system/_apply.py +519 -0
- graph_agents_cli/system/_checks.py +1023 -0
- graph_agents_cli/system/_deploy.py +215 -0
- graph_agents_cli/system/_model.py +363 -0
- graph_agents_cli/system/_system.py +664 -0
- graph_agents_cli/system/_views.py +208 -0
- graph_agents_cli/system/cmd_system.py +423 -0
- graph_agents_cli-0.3.1.dist-info/METADATA +162 -0
- graph_agents_cli-0.3.1.dist-info/RECORD +291 -0
- graph_agents_cli-0.3.1.dist-info/WHEEL +4 -0
- graph_agents_cli-0.3.1.dist-info/entry_points.txt +2 -0
- graph_agents_cli-0.3.1.dist-info/licenses/LICENSE +201 -0
- graph_agents_cli-0.3.1.dist-info/licenses/NOTICE +19 -0
|
@@ -0,0 +1,466 @@
|
|
|
1
|
+
# Copyright 2026 graph-agents-cli contributors
|
|
2
|
+
#
|
|
3
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
4
|
+
# you may not use this file except in compliance with the License.
|
|
5
|
+
# You may obtain a copy of the License at
|
|
6
|
+
#
|
|
7
|
+
# https://www.apache.org/licenses/LICENSE-2.0
|
|
8
|
+
#
|
|
9
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
10
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
11
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
12
|
+
# See the License for the specific language governing permissions and
|
|
13
|
+
# limitations under the License.
|
|
14
|
+
|
|
15
|
+
"""graph-agents-cli eval grade command: score traces and apply the gate."""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import sys
|
|
20
|
+
from pathlib import Path
|
|
21
|
+
from typing import Any
|
|
22
|
+
|
|
23
|
+
import click
|
|
24
|
+
|
|
25
|
+
from graph_agents_cli._output import Console
|
|
26
|
+
from graph_agents_cli._project import find_project_root
|
|
27
|
+
from graph_agents_cli.eval import _paths, gate
|
|
28
|
+
from graph_agents_cli.eval._common import (
|
|
29
|
+
EvalConfigError,
|
|
30
|
+
load_json_file,
|
|
31
|
+
project_meta,
|
|
32
|
+
resolve_judge_identity,
|
|
33
|
+
utc_now_iso,
|
|
34
|
+
write_json_file,
|
|
35
|
+
)
|
|
36
|
+
from graph_agents_cli.eval._judge import results_by_id, run_judge_runner
|
|
37
|
+
from graph_agents_cli.eval.config import EvalConfig, load_eval_config
|
|
38
|
+
from graph_agents_cli.eval.dataset import Dataset, EvalCase, cases_from_raw, load_dataset
|
|
39
|
+
|
|
40
|
+
DEFAULT_JUDGE_TIMEOUT = 600
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _resolve_trace_files(project_root: Path | None, traces_path: str | None) -> list[Path]:
|
|
44
|
+
if traces_path:
|
|
45
|
+
path = Path(traces_path)
|
|
46
|
+
if not path.is_absolute() and project_root is not None:
|
|
47
|
+
candidate = project_root / path
|
|
48
|
+
path = candidate if candidate.exists() or not path.exists() else path
|
|
49
|
+
if path.is_dir():
|
|
50
|
+
files = sorted(path.glob("*.json"))
|
|
51
|
+
if not files:
|
|
52
|
+
raise EvalConfigError(f"no JSON trace files found in {path}")
|
|
53
|
+
return files
|
|
54
|
+
if not path.is_file():
|
|
55
|
+
raise EvalConfigError(f"traces file not found: {path}")
|
|
56
|
+
return [path]
|
|
57
|
+
if project_root is None:
|
|
58
|
+
raise EvalConfigError(
|
|
59
|
+
"not inside a graph-agents-cli project (no manifest found); pass --traces PATH"
|
|
60
|
+
)
|
|
61
|
+
latest = _paths.latest_file(_paths.default_traces_dir(project_root), _paths.TRACES_FILE_PREFIX)
|
|
62
|
+
if latest is None:
|
|
63
|
+
raise EvalConfigError(
|
|
64
|
+
f"no traces found under {_paths.default_traces_dir(project_root)}; "
|
|
65
|
+
"run `graph-agents-cli eval generate` first or pass --traces PATH"
|
|
66
|
+
)
|
|
67
|
+
return [latest]
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def load_traces(files: list[Path]) -> tuple[dict[str, Any], dict[str, dict[str, Any]]]:
|
|
71
|
+
"""Merge trace files into ``(wrapper metadata, traces by case id)``.
|
|
72
|
+
|
|
73
|
+
Every file must come from the same dataset (``dataset_hash``); a case id
|
|
74
|
+
seen twice keeps the entry from the most recently generated file.
|
|
75
|
+
"""
|
|
76
|
+
meta: dict[str, Any] = {"dataset_hash": None, "files": []}
|
|
77
|
+
traces: dict[str, dict[str, Any]] = {}
|
|
78
|
+
generated: dict[str, str] = {}
|
|
79
|
+
embedded: dict[str, dict[str, Any]] = {}
|
|
80
|
+
for path in files:
|
|
81
|
+
data = load_json_file(path, "traces file")
|
|
82
|
+
if not isinstance(data, dict) or not isinstance(data.get("traces"), list):
|
|
83
|
+
raise EvalConfigError(f"traces file {path} must be an object with a 'traces' list")
|
|
84
|
+
digest = data.get("dataset_hash")
|
|
85
|
+
if meta["dataset_hash"] is None:
|
|
86
|
+
meta["dataset_hash"] = digest
|
|
87
|
+
elif digest != meta["dataset_hash"]:
|
|
88
|
+
raise EvalConfigError(
|
|
89
|
+
f"traces file {path} was generated from a different dataset "
|
|
90
|
+
f"({digest} != {meta['dataset_hash']}); grade one dataset at a time"
|
|
91
|
+
)
|
|
92
|
+
for key in (
|
|
93
|
+
"generated_at",
|
|
94
|
+
"agent_version",
|
|
95
|
+
"model",
|
|
96
|
+
"model_provider",
|
|
97
|
+
"dataset_paths",
|
|
98
|
+
"base_url",
|
|
99
|
+
"target",
|
|
100
|
+
):
|
|
101
|
+
if data.get(key) is not None and meta.get(key) is None:
|
|
102
|
+
meta[key] = data[key]
|
|
103
|
+
meta["files"].append(str(path))
|
|
104
|
+
stamp = str(data.get("generated_at") or "")
|
|
105
|
+
for trace in data["traces"]:
|
|
106
|
+
if not isinstance(trace, dict) or not trace.get("case_id"):
|
|
107
|
+
raise EvalConfigError(f"traces file {path} has an entry without case_id")
|
|
108
|
+
case_id = str(trace["case_id"])
|
|
109
|
+
if case_id in traces and generated.get(case_id, "") > stamp:
|
|
110
|
+
continue
|
|
111
|
+
traces[case_id] = trace
|
|
112
|
+
generated[case_id] = stamp
|
|
113
|
+
if isinstance(trace.get("case"), dict):
|
|
114
|
+
embedded[case_id] = trace["case"]
|
|
115
|
+
meta["embedded_cases"] = [embedded[cid] for cid in traces if cid in embedded]
|
|
116
|
+
return meta, traces
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def _planned_dataset(
|
|
120
|
+
project_root: Path | None,
|
|
121
|
+
dataset_flag: str | None,
|
|
122
|
+
meta: dict[str, Any],
|
|
123
|
+
traces: dict[str, dict[str, Any]],
|
|
124
|
+
console: Console,
|
|
125
|
+
) -> Dataset:
|
|
126
|
+
"""The planned cases: ``--dataset``, else the recorded dataset, else embedded cases."""
|
|
127
|
+
root = project_root or Path.cwd()
|
|
128
|
+
if dataset_flag:
|
|
129
|
+
files = _paths.resolve_input_datasets(root, dataset_flag)
|
|
130
|
+
if not files:
|
|
131
|
+
raise EvalConfigError(f"dataset not found: {dataset_flag}")
|
|
132
|
+
dataset = load_dataset(files)
|
|
133
|
+
if meta.get("dataset_hash") and dataset.hash != meta["dataset_hash"]:
|
|
134
|
+
console.print(
|
|
135
|
+
"[yellow]Warning:[/yellow] --dataset differs from the dataset the traces were "
|
|
136
|
+
"generated from (hash mismatch): its expectations are applied to responses "
|
|
137
|
+
"produced for another dataset version, and cases absent from the traces count "
|
|
138
|
+
"as missing. Run `eval generate` to refresh the traces. The results record both "
|
|
139
|
+
"hashes (dataset_hash, traces_dataset_hash)."
|
|
140
|
+
)
|
|
141
|
+
return dataset
|
|
142
|
+
|
|
143
|
+
recorded = meta.get("dataset_paths") or []
|
|
144
|
+
if recorded and project_root is not None:
|
|
145
|
+
files = [project_root / p for p in recorded]
|
|
146
|
+
if all(f.is_file() for f in files):
|
|
147
|
+
try:
|
|
148
|
+
dataset = load_dataset(files)
|
|
149
|
+
except EvalConfigError as exc:
|
|
150
|
+
console.print(f"[yellow]Warning:[/yellow] recorded dataset unusable ({exc}).")
|
|
151
|
+
else:
|
|
152
|
+
if dataset.hash == meta.get("dataset_hash"):
|
|
153
|
+
return dataset
|
|
154
|
+
console.print(
|
|
155
|
+
"[yellow]Warning:[/yellow] the dataset on disk changed since the traces were "
|
|
156
|
+
"generated; grading against the cases embedded in the traces."
|
|
157
|
+
)
|
|
158
|
+
|
|
159
|
+
embedded = meta.get("embedded_cases") or []
|
|
160
|
+
if embedded:
|
|
161
|
+
return cases_from_raw(embedded)
|
|
162
|
+
|
|
163
|
+
console.print(
|
|
164
|
+
"[yellow]Warning:[/yellow] traces carry no case definitions; pass --dataset to apply "
|
|
165
|
+
"expectations. Grading case accounting only."
|
|
166
|
+
)
|
|
167
|
+
cases = [EvalCase(id=cid, messages=[{"role": "user", "content": ""}]) for cid in traces]
|
|
168
|
+
return Dataset(cases=cases, hash=str(meta.get("dataset_hash") or ""))
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
FAKE_PROVIDER = "fake"
|
|
172
|
+
|
|
173
|
+
FAKE_JUDGE_WARNING = (
|
|
174
|
+
"the judge is the deterministic fake model (provider 'fake'): it gives every judge metric "
|
|
175
|
+
"the maximum score without reading the reply, so the judge scores and quality rates are "
|
|
176
|
+
"not a quality signal. Set JUDGE_MODEL_PROVIDER (or judge.provider in eval_config.yaml) to "
|
|
177
|
+
"a real provider to measure quality."
|
|
178
|
+
)
|
|
179
|
+
FAKE_AGENT_WARNING = (
|
|
180
|
+
"the agent ran on the deterministic fake model (MODEL_PROVIDER=fake): its replies are "
|
|
181
|
+
"canned, so this run proves the eval plumbing only, not the agent's behaviour. Run it on "
|
|
182
|
+
"the project's real provider before trusting the gate."
|
|
183
|
+
)
|
|
184
|
+
|
|
185
|
+
FAKE_URL_AGENT_WARNING = (
|
|
186
|
+
"this project's settings name the deterministic fake model (MODEL_PROVIDER=fake). The "
|
|
187
|
+
"agent at the --url target may run another model, but if it runs these settings (a local "
|
|
188
|
+
"server started from this project, say), its replies are canned and the gate proves the "
|
|
189
|
+
"eval plumbing only."
|
|
190
|
+
)
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def _is_fake(provider: Any) -> bool:
|
|
194
|
+
return str(provider or "").strip().lower() == FAKE_PROVIDER
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def _fake_models(
|
|
198
|
+
meta: dict[str, Any], judge: dict[str, Any], items: list[dict[str, Any]]
|
|
199
|
+
) -> list[str]:
|
|
200
|
+
"""``agent`` and/or ``judge`` when that side ran on the fake model.
|
|
201
|
+
|
|
202
|
+
The agent's provider is known only for traces from the project's own local
|
|
203
|
+
server (``target: local``); a ``--url`` agent may run any model. The judge
|
|
204
|
+
counts only when a model judge actually scored something.
|
|
205
|
+
"""
|
|
206
|
+
fake: list[str] = []
|
|
207
|
+
if meta.get("target") == "local" and _is_fake(meta.get("model_provider")):
|
|
208
|
+
fake.append("agent")
|
|
209
|
+
if any(item.get("kind") == "judge" for item in items) and _is_fake(judge.get("provider")):
|
|
210
|
+
fake.append("judge")
|
|
211
|
+
return fake
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
def _grade_warnings(
|
|
215
|
+
config: EvalConfig,
|
|
216
|
+
grades: list[gate.CaseGrade],
|
|
217
|
+
fake_model: list[str],
|
|
218
|
+
meta: dict[str, Any] | None = None,
|
|
219
|
+
) -> list[str]:
|
|
220
|
+
warnings: list[str] = []
|
|
221
|
+
if "agent" in fake_model:
|
|
222
|
+
warnings.append(FAKE_AGENT_WARNING)
|
|
223
|
+
elif (meta or {}).get("target") == "url" and _is_fake((meta or {}).get("model_provider")):
|
|
224
|
+
warnings.append(FAKE_URL_AGENT_WARNING)
|
|
225
|
+
if "judge" in fake_model:
|
|
226
|
+
warnings.append(FAKE_JUDGE_WARNING)
|
|
227
|
+
cut = [g for g in grades if g.truncated_tool_results]
|
|
228
|
+
if cut:
|
|
229
|
+
total = sum(g.truncated_tool_results for g in cut)
|
|
230
|
+
where = config.source.name if config.source else "eval_config.yaml"
|
|
231
|
+
warnings.append(
|
|
232
|
+
f"{total} tool result(s) in {len(cut)} case(s) were longer than "
|
|
233
|
+
f"judge.max_tool_result_chars ({config.max_tool_result_chars}) and were cut in the "
|
|
234
|
+
"judge prompt, marked TRUNCATED (the judge is told not to count the omitted part as "
|
|
235
|
+
f"unsupported): {', '.join(g.id for g in cut[:5])}{'...' if len(cut) > 5 else ''}. "
|
|
236
|
+
f"Raise judge.max_tool_result_chars in {where}, or set it to null, to show them in full."
|
|
237
|
+
)
|
|
238
|
+
final_only = [g.id for g in grades if g.checks_final_turn_only]
|
|
239
|
+
if final_only:
|
|
240
|
+
warnings.append(
|
|
241
|
+
f"{len(final_only)} case(s) declare expect.scope: all_turns but their traces have no "
|
|
242
|
+
"per-turn records (an older traces file or an eval generate override), so their "
|
|
243
|
+
"checks read the final turn only: "
|
|
244
|
+
f"{', '.join(final_only[:5])}{'...' if len(final_only) > 5 else ''}. Re-run "
|
|
245
|
+
"`eval generate` to record every turn."
|
|
246
|
+
)
|
|
247
|
+
unrecorded = [g.id for g in grades if g.unrecorded_turns]
|
|
248
|
+
if unrecorded:
|
|
249
|
+
warnings.append(
|
|
250
|
+
f"{len(unrecorded)} multi-turn case(s) have traces without per-turn records (an older "
|
|
251
|
+
"traces file or an eval generate override), so their judges saw the earlier user "
|
|
252
|
+
"messages but not the agent's earlier replies: "
|
|
253
|
+
f"{', '.join(unrecorded[:5])}{'...' if len(unrecorded) > 5 else ''}. Re-run "
|
|
254
|
+
"`eval generate` to record every turn."
|
|
255
|
+
)
|
|
256
|
+
return warnings
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
def grade_traces(
|
|
260
|
+
*,
|
|
261
|
+
traces_path: str | None = None,
|
|
262
|
+
dataset: str | None = None,
|
|
263
|
+
config_path: str | None = None,
|
|
264
|
+
output_path: str | None = None,
|
|
265
|
+
judge_provider: str | None = None,
|
|
266
|
+
judge_model: str | None = None,
|
|
267
|
+
judge_timeout: int = DEFAULT_JUDGE_TIMEOUT,
|
|
268
|
+
console: Console | None = None,
|
|
269
|
+
) -> int:
|
|
270
|
+
"""Grade traces and return the gate exit code (0/1/2). Raises EvalConfigError for 3."""
|
|
271
|
+
console = console or Console()
|
|
272
|
+
project_root = find_project_root()
|
|
273
|
+
|
|
274
|
+
files = _resolve_trace_files(project_root, traces_path)
|
|
275
|
+
console.print(f"Loading traces from [cyan]{', '.join(str(f) for f in files)}[/cyan]")
|
|
276
|
+
meta, traces = load_traces(files)
|
|
277
|
+
|
|
278
|
+
if config_path:
|
|
279
|
+
cfg_file = Path(config_path)
|
|
280
|
+
if not cfg_file.is_absolute() and project_root is not None and not cfg_file.exists():
|
|
281
|
+
cfg_file = project_root / cfg_file
|
|
282
|
+
config: EvalConfig = load_eval_config(cfg_file, required=True)
|
|
283
|
+
else:
|
|
284
|
+
config = load_eval_config(
|
|
285
|
+
_paths.default_eval_config(project_root) if project_root else None
|
|
286
|
+
)
|
|
287
|
+
|
|
288
|
+
planned = _planned_dataset(project_root, dataset, meta, traces, console)
|
|
289
|
+
for case in planned.cases:
|
|
290
|
+
gate.validate_case_metrics(config, case)
|
|
291
|
+
|
|
292
|
+
extra = sorted(set(traces) - set(planned.case_ids))
|
|
293
|
+
if extra:
|
|
294
|
+
console.print(
|
|
295
|
+
f"[yellow]Warning:[/yellow] {len(extra)} trace(s) not in the dataset are ignored: "
|
|
296
|
+
f"{', '.join(extra[:5])}{'...' if len(extra) > 5 else ''}"
|
|
297
|
+
)
|
|
298
|
+
|
|
299
|
+
grades: list[gate.CaseGrade] = []
|
|
300
|
+
items: list[dict[str, Any]] = []
|
|
301
|
+
for case in planned.cases:
|
|
302
|
+
trace = traces.get(case.id)
|
|
303
|
+
grade = gate.grade_deterministic(case, trace)
|
|
304
|
+
if trace is not None and grade.judgeable:
|
|
305
|
+
items.extend(gate.plan_judge_items(config, case, trace, grade))
|
|
306
|
+
grades.append(grade)
|
|
307
|
+
|
|
308
|
+
judge = resolve_judge_identity(
|
|
309
|
+
project_root,
|
|
310
|
+
config_provider=config.judge_provider,
|
|
311
|
+
config_model=config.judge_model,
|
|
312
|
+
flag_provider=judge_provider,
|
|
313
|
+
flag_model=judge_model,
|
|
314
|
+
)
|
|
315
|
+
if items:
|
|
316
|
+
if project_root is None:
|
|
317
|
+
raise EvalConfigError(
|
|
318
|
+
"judge metrics need the project's environment; run `eval grade` inside the "
|
|
319
|
+
"project (no graph-agents-cli-manifest.yaml found)"
|
|
320
|
+
)
|
|
321
|
+
console.print(
|
|
322
|
+
f"Running {len(items)} judge/custom metric call(s) in the project environment "
|
|
323
|
+
f"(judge: {judge['provider'] or 'default'}/{judge['model'] or 'default'})..."
|
|
324
|
+
)
|
|
325
|
+
output = run_judge_runner(
|
|
326
|
+
project_root, {"judge": judge, "items": items}, timeout=judge_timeout
|
|
327
|
+
)
|
|
328
|
+
results = results_by_id(output)
|
|
329
|
+
for grade in grades:
|
|
330
|
+
if grade.judge_scores:
|
|
331
|
+
gate.apply_judge_results(grade, results)
|
|
332
|
+
judge = {
|
|
333
|
+
"provider": output.get("provider") or judge["provider"],
|
|
334
|
+
"model": output.get("model") or judge["model"],
|
|
335
|
+
}
|
|
336
|
+
|
|
337
|
+
summary = gate.summarize(grades)
|
|
338
|
+
quality = gate.compute_quality(config, grades)
|
|
339
|
+
exit_code = gate.exit_code_for(summary, quality)
|
|
340
|
+
summary["exit_code"] = exit_code
|
|
341
|
+
identity = project_meta(project_root)
|
|
342
|
+
fake_model = _fake_models(meta, judge, items)
|
|
343
|
+
warnings = _grade_warnings(config, grades, fake_model, meta)
|
|
344
|
+
|
|
345
|
+
results_doc = {
|
|
346
|
+
"dataset_hash": planned.hash or meta.get("dataset_hash"),
|
|
347
|
+
# Provenance (additive fields): the dataset the traces were
|
|
348
|
+
# generated from, which differs from dataset_hash after `--dataset`.
|
|
349
|
+
"traces_dataset_hash": meta.get("dataset_hash"),
|
|
350
|
+
"graded_at": utc_now_iso(),
|
|
351
|
+
"generated_at": meta.get("generated_at"),
|
|
352
|
+
"judge": judge,
|
|
353
|
+
"capture": identity["capture"],
|
|
354
|
+
"agent_version": meta.get("agent_version") or identity["agent_version"],
|
|
355
|
+
# A --url agent's model is unknown (null): never the project's settings.
|
|
356
|
+
"model": meta.get("model")
|
|
357
|
+
if meta.get("target") == "url"
|
|
358
|
+
else meta.get("model") or identity["model"],
|
|
359
|
+
"traces_files": meta["files"],
|
|
360
|
+
"dataset_paths": [str(p) for p in planned.sources] or meta.get("dataset_paths") or [],
|
|
361
|
+
"config": str(config.source) if config.source else None,
|
|
362
|
+
"planned": len(grades),
|
|
363
|
+
"summary": summary,
|
|
364
|
+
"quality": quality,
|
|
365
|
+
# Which of the agent and the judge ran on the deterministic fake model:
|
|
366
|
+
# a gate met that way proves the plumbing only.
|
|
367
|
+
"fake_model": fake_model,
|
|
368
|
+
"warnings": warnings,
|
|
369
|
+
"cases": [g.to_dict() for g in grades],
|
|
370
|
+
}
|
|
371
|
+
|
|
372
|
+
root_for_output = project_root or Path.cwd()
|
|
373
|
+
out = _paths.resolve_output_path(
|
|
374
|
+
root_for_output,
|
|
375
|
+
output_path,
|
|
376
|
+
default_dir=_paths.default_grade_results_dir(root_for_output),
|
|
377
|
+
prefix=_paths.RESULTS_FILE_PREFIX,
|
|
378
|
+
)
|
|
379
|
+
write_json_file(out, results_doc)
|
|
380
|
+
gate.print_summary(console, results_doc)
|
|
381
|
+
console.print(f"Results saved to [green]{out}[/green]")
|
|
382
|
+
return exit_code
|
|
383
|
+
|
|
384
|
+
|
|
385
|
+
@click.command("grade")
|
|
386
|
+
@click.option(
|
|
387
|
+
"--traces",
|
|
388
|
+
"traces_path",
|
|
389
|
+
type=click.Path(),
|
|
390
|
+
default=None,
|
|
391
|
+
help=(
|
|
392
|
+
"Traces file, or a directory whose *.json files are merged (all from one dataset). "
|
|
393
|
+
"Defaults to the newest artifacts/traces/traces_<ts>.json."
|
|
394
|
+
),
|
|
395
|
+
)
|
|
396
|
+
@click.option(
|
|
397
|
+
"--dataset",
|
|
398
|
+
default=None,
|
|
399
|
+
help=(
|
|
400
|
+
"Dataset file or directory to re-read for planned-case accounting and expectations. "
|
|
401
|
+
"Defaults to the dataset recorded in the traces (falling back to the cases embedded in them)."
|
|
402
|
+
),
|
|
403
|
+
)
|
|
404
|
+
@click.option(
|
|
405
|
+
"--config",
|
|
406
|
+
"config_path",
|
|
407
|
+
type=click.Path(),
|
|
408
|
+
default=None,
|
|
409
|
+
help=f"Eval config (judge, quality_metrics, judges, custom_metrics). Defaults to {_paths.DEFAULT_EVAL_CONFIG}.",
|
|
410
|
+
)
|
|
411
|
+
@click.option(
|
|
412
|
+
"--output",
|
|
413
|
+
"-o",
|
|
414
|
+
"output_path",
|
|
415
|
+
default=None,
|
|
416
|
+
help=(
|
|
417
|
+
"Results file, or a directory to write results_<ts>.json into. Defaults to "
|
|
418
|
+
f"{_paths.ARTIFACTS_DIR}/{_paths.GRADE_RESULTS_SUBDIR}/results_<ts>.json."
|
|
419
|
+
),
|
|
420
|
+
)
|
|
421
|
+
@click.option(
|
|
422
|
+
"--judge-provider", default=None, help="Override the judge model provider for this run."
|
|
423
|
+
)
|
|
424
|
+
@click.option("--judge-model", default=None, help="Override the judge model name for this run.")
|
|
425
|
+
@click.option(
|
|
426
|
+
"--judge-timeout",
|
|
427
|
+
type=click.IntRange(min=1),
|
|
428
|
+
default=DEFAULT_JUDGE_TIMEOUT,
|
|
429
|
+
show_default=True,
|
|
430
|
+
help="Seconds allowed for the judge runner (all judge calls of the run).",
|
|
431
|
+
)
|
|
432
|
+
def cmd_grade(
|
|
433
|
+
*,
|
|
434
|
+
traces_path: str | None,
|
|
435
|
+
dataset: str | None,
|
|
436
|
+
config_path: str | None,
|
|
437
|
+
output_path: str | None,
|
|
438
|
+
judge_provider: str | None,
|
|
439
|
+
judge_model: str | None,
|
|
440
|
+
judge_timeout: int,
|
|
441
|
+
) -> None:
|
|
442
|
+
"""Grade traces against the dataset's checks and judges and apply the gate.
|
|
443
|
+
|
|
444
|
+
Deterministic checks run in-process first; judge metrics and custom metrics
|
|
445
|
+
run inside the project's environment through the staged judge runner.
|
|
446
|
+
Every planned case ends as passed, failed, quality_below_threshold, error
|
|
447
|
+
or missing, and the results file records the per-status counts.
|
|
448
|
+
|
|
449
|
+
\b
|
|
450
|
+
Exit codes:
|
|
451
|
+
0 gate met
|
|
452
|
+
1 a case failed, or a quality metric is under its min_pass_rate
|
|
453
|
+
2 a case is error or missing (incomplete run)
|
|
454
|
+
3 configuration error (unknown metric, unreachable judge, no threshold)
|
|
455
|
+
"""
|
|
456
|
+
code = grade_traces(
|
|
457
|
+
traces_path=traces_path,
|
|
458
|
+
dataset=dataset,
|
|
459
|
+
config_path=config_path,
|
|
460
|
+
output_path=output_path,
|
|
461
|
+
judge_provider=judge_provider,
|
|
462
|
+
judge_model=judge_model,
|
|
463
|
+
judge_timeout=judge_timeout,
|
|
464
|
+
)
|
|
465
|
+
if code:
|
|
466
|
+
sys.exit(code)
|
|
@@ -0,0 +1,156 @@
|
|
|
1
|
+
# Copyright 2026 graph-agents-cli contributors
|
|
2
|
+
#
|
|
3
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
4
|
+
# you may not use this file except in compliance with the License.
|
|
5
|
+
# You may obtain a copy of the License at
|
|
6
|
+
#
|
|
7
|
+
# https://www.apache.org/licenses/LICENSE-2.0
|
|
8
|
+
#
|
|
9
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
10
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
11
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
12
|
+
# See the License for the specific language governing permissions and
|
|
13
|
+
# limitations under the License.
|
|
14
|
+
|
|
15
|
+
"""graph-agents-cli eval metric commands."""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import click
|
|
20
|
+
from rich.table import Table
|
|
21
|
+
|
|
22
|
+
from graph_agents_cli._click import LazyGroup
|
|
23
|
+
from graph_agents_cli._output import Console, emit
|
|
24
|
+
from graph_agents_cli._project import find_project_root
|
|
25
|
+
from graph_agents_cli.eval import _paths
|
|
26
|
+
from graph_agents_cli.eval.checks import CHECK_DESCRIPTIONS, MODIFIER_DESCRIPTIONS
|
|
27
|
+
from graph_agents_cli.eval.config import BUILTIN_JUDGES, EvalConfig, load_eval_config
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@click.group("metric", cls=LazyGroup)
|
|
31
|
+
def metric_group() -> None:
|
|
32
|
+
"""Discover evaluation metrics: deterministic checks and judges."""
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _project_config() -> EvalConfig | None:
|
|
36
|
+
project_root = find_project_root()
|
|
37
|
+
if project_root is None:
|
|
38
|
+
return None
|
|
39
|
+
path = _paths.default_eval_config(project_root)
|
|
40
|
+
if not path.is_file():
|
|
41
|
+
return None
|
|
42
|
+
try:
|
|
43
|
+
return load_eval_config(path)
|
|
44
|
+
except click.ClickException:
|
|
45
|
+
return None
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
@click.command("list")
|
|
49
|
+
@click.option("--json", "as_json", is_flag=True, help="Print the catalogue as JSON.")
|
|
50
|
+
def list_metrics(*, as_json: bool) -> None:
|
|
51
|
+
"""List deterministic checks and built-in judges (plus this project's config)."""
|
|
52
|
+
config = _project_config()
|
|
53
|
+
judges = []
|
|
54
|
+
for name, spec in BUILTIN_JUDGES.items():
|
|
55
|
+
judges.append(
|
|
56
|
+
{
|
|
57
|
+
"name": name,
|
|
58
|
+
"scale": spec["scale"],
|
|
59
|
+
"description": spec["description"],
|
|
60
|
+
"source": "built-in",
|
|
61
|
+
}
|
|
62
|
+
)
|
|
63
|
+
custom = []
|
|
64
|
+
if config is not None:
|
|
65
|
+
for name, spec in config.judges.items():
|
|
66
|
+
if not spec.builtin:
|
|
67
|
+
judges.append(
|
|
68
|
+
{
|
|
69
|
+
"name": name,
|
|
70
|
+
"scale": spec.scale,
|
|
71
|
+
"description": spec.description or "custom rubric",
|
|
72
|
+
"source": str(config.source),
|
|
73
|
+
}
|
|
74
|
+
)
|
|
75
|
+
for name, metric in config.custom_metrics.items():
|
|
76
|
+
custom.append(
|
|
77
|
+
{
|
|
78
|
+
"name": name,
|
|
79
|
+
"callable": metric.callable,
|
|
80
|
+
"threshold": metric.threshold,
|
|
81
|
+
"description": metric.description,
|
|
82
|
+
"source": str(config.source),
|
|
83
|
+
}
|
|
84
|
+
)
|
|
85
|
+
quality = sorted(config.quality_metrics)
|
|
86
|
+
else:
|
|
87
|
+
quality = []
|
|
88
|
+
|
|
89
|
+
catalogue = {
|
|
90
|
+
"checks": [{"name": n, "description": d} for n, d in CHECK_DESCRIPTIONS.items()],
|
|
91
|
+
"modifiers": [{"name": n, "description": d} for n, d in MODIFIER_DESCRIPTIONS.items()],
|
|
92
|
+
"judges": judges,
|
|
93
|
+
"custom_metrics": custom,
|
|
94
|
+
"quality_metrics": quality,
|
|
95
|
+
}
|
|
96
|
+
if as_json:
|
|
97
|
+
emit(catalogue)
|
|
98
|
+
return
|
|
99
|
+
|
|
100
|
+
console = Console()
|
|
101
|
+
checks = Table(
|
|
102
|
+
title="Deterministic checks (expect.<name>)", show_header=True, header_style="bold"
|
|
103
|
+
)
|
|
104
|
+
checks.add_column("Check", style="cyan", no_wrap=True)
|
|
105
|
+
checks.add_column("Description")
|
|
106
|
+
for entry in catalogue["checks"]:
|
|
107
|
+
checks.add_row(entry["name"], entry["description"])
|
|
108
|
+
console.print(checks)
|
|
109
|
+
|
|
110
|
+
modifiers = Table(
|
|
111
|
+
title="Check modifiers (expect.<name>, not checks)", show_header=True, header_style="bold"
|
|
112
|
+
)
|
|
113
|
+
modifiers.add_column("Modifier", style="cyan", no_wrap=True)
|
|
114
|
+
modifiers.add_column("Description")
|
|
115
|
+
for entry in catalogue["modifiers"]:
|
|
116
|
+
modifiers.add_row(entry["name"], entry["description"])
|
|
117
|
+
console.print(modifiers)
|
|
118
|
+
|
|
119
|
+
table = Table(title="Judge metrics (judge.<name>)", show_header=True, header_style="bold")
|
|
120
|
+
table.add_column("Judge", style="cyan", no_wrap=True)
|
|
121
|
+
table.add_column("Scale", justify="right")
|
|
122
|
+
table.add_column("Description")
|
|
123
|
+
table.add_column("Source")
|
|
124
|
+
for entry in judges:
|
|
125
|
+
mark = " [yellow](quality)[/yellow]" if entry["name"] in quality else ""
|
|
126
|
+
table.add_row(
|
|
127
|
+
entry["name"] + mark, str(entry["scale"]), entry["description"], entry["source"]
|
|
128
|
+
)
|
|
129
|
+
console.print(table)
|
|
130
|
+
|
|
131
|
+
if custom:
|
|
132
|
+
ctable = Table(
|
|
133
|
+
title="Custom metrics (custom_metrics in eval_config.yaml)",
|
|
134
|
+
show_header=True,
|
|
135
|
+
header_style="bold",
|
|
136
|
+
)
|
|
137
|
+
ctable.add_column("Metric", style="cyan", no_wrap=True)
|
|
138
|
+
ctable.add_column("Callable")
|
|
139
|
+
ctable.add_column("Threshold", justify="right")
|
|
140
|
+
for entry in custom:
|
|
141
|
+
mark = " [yellow](quality)[/yellow]" if entry["name"] in quality else ""
|
|
142
|
+
ctable.add_row(entry["name"] + mark, entry["callable"], str(entry["threshold"]))
|
|
143
|
+
console.print(ctable)
|
|
144
|
+
|
|
145
|
+
console.print(
|
|
146
|
+
"Judge metrics are mandatory unless listed under quality_metrics in "
|
|
147
|
+
f"{_paths.DEFAULT_EVAL_CONFIG}; custom metrics are module:function callables "
|
|
148
|
+
"run inside the project's environment."
|
|
149
|
+
)
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
metric_group.add_lazy_command(
|
|
153
|
+
"list",
|
|
154
|
+
"graph_agents_cli.eval.cmd_metric:list_metrics",
|
|
155
|
+
"List deterministic checks and built-in judges (plus this project's config).",
|
|
156
|
+
)
|