graph-agents-cli 0.3.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- graph_agents_cli/__init__.py +26 -0
- graph_agents_cli/_api_policy.py +2145 -0
- graph_agents_cli/_approvals.py +400 -0
- graph_agents_cli/_build.py +186 -0
- graph_agents_cli/_build_info.json +7 -0
- graph_agents_cli/_chat_client.py +462 -0
- graph_agents_cli/_click.py +157 -0
- graph_agents_cli/_defaults.py +139 -0
- graph_agents_cli/_experiments.py +64 -0
- graph_agents_cli/_http.py +192 -0
- graph_agents_cli/_output.py +83 -0
- graph_agents_cli/_project.py +462 -0
- graph_agents_cli/_remote.py +220 -0
- graph_agents_cli/_response_schema.py +264 -0
- graph_agents_cli/_runner.py +319 -0
- graph_agents_cli/_skills_check.py +274 -0
- graph_agents_cli/_tools.py +189 -0
- graph_agents_cli/_trust.py +66 -0
- graph_agents_cli/api/__init__.py +15 -0
- graph_agents_cli/api/_changes.py +506 -0
- graph_agents_cli/api/_files.py +658 -0
- graph_agents_cli/api/cmd_api.py +2480 -0
- graph_agents_cli/deploy/__init__.py +15 -0
- graph_agents_cli/deploy/_config.py +171 -0
- graph_agents_cli/deploy/_image.py +128 -0
- graph_agents_cli/deploy/_kube.py +286 -0
- graph_agents_cli/deploy/_modes.py +234 -0
- graph_agents_cli/deploy/_preflight.py +370 -0
- graph_agents_cli/deploy/_values.py +168 -0
- graph_agents_cli/deploy/cmd_deploy.py +1866 -0
- graph_agents_cli/deploy/gitops.py +562 -0
- graph_agents_cli/deploy/local_load.py +273 -0
- graph_agents_cli/dev/__init__.py +13 -0
- graph_agents_cli/dev/cmd_build.py +131 -0
- graph_agents_cli/dev/cmd_install.py +78 -0
- graph_agents_cli/dev/cmd_lint.py +119 -0
- graph_agents_cli/dev/cmd_playground.py +297 -0
- graph_agents_cli/dev/policy_check.py +1287 -0
- graph_agents_cli/eval/__init__.py +22 -0
- graph_agents_cli/eval/_client.py +670 -0
- graph_agents_cli/eval/_common.py +177 -0
- graph_agents_cli/eval/_judge.py +168 -0
- graph_agents_cli/eval/_judge_runner.py +238 -0
- graph_agents_cli/eval/_paths.py +212 -0
- graph_agents_cli/eval/checks.py +581 -0
- graph_agents_cli/eval/cmd_analyze.py +278 -0
- graph_agents_cli/eval/cmd_compare.py +284 -0
- graph_agents_cli/eval/cmd_eval_group.py +80 -0
- graph_agents_cli/eval/cmd_generate.py +558 -0
- graph_agents_cli/eval/cmd_grade.py +466 -0
- graph_agents_cli/eval/cmd_metric.py +156 -0
- graph_agents_cli/eval/cmd_run.py +370 -0
- graph_agents_cli/eval/cmd_submit.py +400 -0
- graph_agents_cli/eval/config.py +435 -0
- graph_agents_cli/eval/dataset.py +350 -0
- graph_agents_cli/eval/gate.py +420 -0
- graph_agents_cli/eval/transcript.py +192 -0
- graph_agents_cli/extension/__init__.py +13 -0
- graph_agents_cli/extension/_compat.py +86 -0
- graph_agents_cli/extension/_loader.py +293 -0
- graph_agents_cli/extension/_manifest.py +135 -0
- graph_agents_cli/extension/_overrides.py +195 -0
- graph_agents_cli/extension/_paths.py +91 -0
- graph_agents_cli/extension/_refs.py +193 -0
- graph_agents_cli/extension/_resolver.py +453 -0
- graph_agents_cli/extension/_schema.py +106 -0
- graph_agents_cli/extension/_spec.py +253 -0
- graph_agents_cli/extension/_sync.py +102 -0
- graph_agents_cli/extension/_trust.py +58 -0
- graph_agents_cli/extension/cmd_extension_add.py +259 -0
- graph_agents_cli/extension/cmd_extension_group.py +57 -0
- graph_agents_cli/extension/cmd_extension_list.py +56 -0
- graph_agents_cli/extension/cmd_extension_remove.py +61 -0
- graph_agents_cli/extension/cmd_extension_update.py +195 -0
- graph_agents_cli/info/__init__.py +13 -0
- graph_agents_cli/info/cmd_info.py +222 -0
- graph_agents_cli/infra/__init__.py +15 -0
- graph_agents_cli/infra/checks.py +1169 -0
- graph_agents_cli/infra/cmd_infra.py +103 -0
- graph_agents_cli/main.py +591 -0
- graph_agents_cli/peer/__init__.py +15 -0
- graph_agents_cli/peer/_generate.py +254 -0
- graph_agents_cli/peer/cmd_peer.py +1151 -0
- graph_agents_cli/run/__init__.py +13 -0
- graph_agents_cli/run/_local_server.py +1157 -0
- graph_agents_cli/run/_signals.py +141 -0
- graph_agents_cli/run/cmd_approvals.py +530 -0
- graph_agents_cli/run/cmd_run.py +1421 -0
- graph_agents_cli/scaffold/__init__.py +19 -0
- graph_agents_cli/scaffold/agents/README.md +24 -0
- graph_agents_cli/scaffold/agents/empty_py/.template/templateconfig.yaml +22 -0
- graph_agents_cli/scaffold/agents/langgraph/.env.example +292 -0
- graph_agents_cli/scaffold/agents/langgraph/.template/templateconfig.yaml +28 -0
- graph_agents_cli/scaffold/agents/langgraph/Dockerfile +59 -0
- graph_agents_cli/scaffold/agents/langgraph/Dockerfile.langgraph-server +59 -0
- graph_agents_cli/scaffold/agents/langgraph/README.md +571 -0
- graph_agents_cli/scaffold/agents/langgraph/api-policy.yaml +60 -0
- graph_agents_cli/scaffold/agents/langgraph/app/__init__.py +20 -0
- graph_agents_cli/scaffold/agents/langgraph/app/agent.py +174 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/__init__.py +15 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/a2a.py +2162 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/a2a_client.py +1167 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/api_client.py +4220 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/approvals.py +1349 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/auth.py +1986 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/chat.py +2962 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/checkpointer.py +432 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/content.py +569 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/db.py +580 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/limits.py +203 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/metrics.py +231 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/middleware.py +361 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/model.py +611 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/playground.py +230 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/run_locks.py +459 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/structured.py +755 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/telemetry.py +681 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/threads.py +493 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/token_exchange.py +959 -0
- graph_agents_cli/scaffold/agents/langgraph/app/fast_api_app.py +770 -0
- graph_agents_cli/scaffold/agents/langgraph/app/policies/__init__.py +55 -0
- graph_agents_cli/scaffold/agents/langgraph/app/policies/custom.py +97 -0
- graph_agents_cli/scaffold/agents/langgraph/app/tools/__init__.py +46 -0
- graph_agents_cli/scaffold/agents/langgraph/app/tools/example_api.py +92 -0
- graph_agents_cli/scaffold/agents/langgraph/app/tools/weather.py +33 -0
- graph_agents_cli/scaffold/agents/langgraph/langgraph.json +14 -0
- graph_agents_cli/scaffold/agents/langgraph/pyproject.toml +78 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/conftest.py +376 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/eval/datasets/basic-dataset.json +53 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/eval/eval_config.yaml +32 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/approval_graph.py +137 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/fake_issuer.py +216 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/fake_openai.py +357 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_a2a_outcomes.py +569 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_a2a_relay.py +479 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_api_surface.py +812 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_approvals.py +1367 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_approvals_server.py +794 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_cross_actor.py +497 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_cross_actor_server.py +247 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_history_repair.py +278 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_model_apis.py +242 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_postgres.py +637 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_resilience_postgres.py +770 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_runtime_guardrails.py +854 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_server_e2e.py +340 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_server_runtime.py +989 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_structured_answers.py +584 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_structured_server.py +222 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_token_exchange_issuer.py +650 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/load_test/.results/.placeholder +0 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/load_test/README.md +22 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/load_test/conftest.py +21 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/load_test/load_test.py +81 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_a2a_client.py +824 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_a2a_scoping.py +724 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_api_client.py +1214 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_api_client_hardening.py +716 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_api_policy_rpc.py +767 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_approval_ledger.py +1536 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_fake_model.py +115 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_jwt_policy.py +991 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_limits.py +310 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_logging.py +148 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_logging_hardening.py +271 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_policy.py +378 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_resilience.py +610 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_server_auth.py +702 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_structured.py +673 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_telemetry.py +404 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_thread_listing.py +255 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_threads.py +268 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_token_exchange.py +1320 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_untrusted_content.py +393 -0
- graph_agents_cli/scaffold/agents/langgraph/uv-fastapi.lock +2084 -0
- graph_agents_cli/scaffold/agents/langgraph/uv-langgraph-server.lock +2106 -0
- graph_agents_cli/scaffold/agents/langgraph/{{cookiecutter.agent_guidance_filename}} +129 -0
- graph_agents_cli/scaffold/base_templates/_shared/graph-agents-cli-manifest.yaml +36 -0
- graph_agents_cli/scaffold/base_templates/python/.dockerignore +32 -0
- graph_agents_cli/scaffold/base_templates/python/.github/CODEOWNERS +30 -0
- graph_agents_cli/scaffold/base_templates/python/.github/agent.env +7 -0
- graph_agents_cli/scaffold/base_templates/python/.github/workflows/pr_checks.yaml +214 -0
- graph_agents_cli/scaffold/base_templates/python/.gitignore +209 -0
- graph_agents_cli/scaffold/base_templates/python/tests/unit/test_dummy.py +23 -0
- graph_agents_cli/scaffold/base_templates/python/{{cookiecutter.agent_guidance_filename}} +35 -0
- graph_agents_cli/scaffold/cmd_scaffold_group.py +49 -0
- graph_agents_cli/scaffold/commands/__init__.py +13 -0
- graph_agents_cli/scaffold/commands/create.py +1424 -0
- graph_agents_cli/scaffold/commands/enhance.py +1652 -0
- graph_agents_cli/scaffold/commands/upgrade.py +570 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/.github/agent.env +12 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/.github/workflows/promote-to-prod.yaml +371 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/.github/workflows/staging.yaml +450 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/argocd/application-dev.yaml +43 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/argocd/application-prod.yaml +41 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/argocd/application-staging.yaml +43 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/.helmignore +14 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/Chart.yaml +21 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/examples/networkpolicy.yaml +103 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/NOTES.txt +48 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/_helpers.tpl +189 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/certificate.yaml +15 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/configmap.yaml +10 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/deployment.yaml +199 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/hpa.yaml +22 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/httproute.yaml +30 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/ingress.yaml +39 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/networkpolicy.yaml +48 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/pdb.yaml +13 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/postgresql-secret.yaml +37 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/service.yaml +15 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/serviceaccount.yaml +13 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/servicemonitor.yaml +42 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values-dev.yaml +22 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values-prod.yaml +45 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values-staging.yaml +29 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values.yaml +396 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/tests/integration/test_chart.py +269 -0
- graph_agents_cli/scaffold/deployment_targets/none/README.md +5 -0
- graph_agents_cli/scaffold/deployment_targets/none/python/README.md +6 -0
- graph_agents_cli/scaffold/utils/__init__.py +13 -0
- graph_agents_cli/scaffold/utils/backup.py +212 -0
- graph_agents_cli/scaffold/utils/build_record.py +257 -0
- graph_agents_cli/scaffold/utils/cli_options.py +184 -0
- graph_agents_cli/scaffold/utils/fs.py +83 -0
- graph_agents_cli/scaffold/utils/generate_locks.py +214 -0
- graph_agents_cli/scaffold/utils/generation_metadata.py +88 -0
- graph_agents_cli/scaffold/utils/keyedit.py +768 -0
- graph_agents_cli/scaffold/utils/keymerge.py +537 -0
- graph_agents_cli/scaffold/utils/language.py +138 -0
- graph_agents_cli/scaffold/utils/lock_utils.py +94 -0
- graph_agents_cli/scaffold/utils/logging.py +77 -0
- graph_agents_cli/scaffold/utils/manifest.py +292 -0
- graph_agents_cli/scaffold/utils/merge.py +970 -0
- graph_agents_cli/scaffold/utils/merge3.py +216 -0
- graph_agents_cli/scaffold/utils/openapi_seed.py +199 -0
- graph_agents_cli/scaffold/utils/remote_template.py +376 -0
- graph_agents_cli/scaffold/utils/template.py +1352 -0
- graph_agents_cli/scaffold/utils/upgrade.py +894 -0
- graph_agents_cli/scaffold/utils/version.py +438 -0
- graph_agents_cli/secrets/__init__.py +15 -0
- graph_agents_cli/secrets/_apply.py +954 -0
- graph_agents_cli/secrets/_required.py +188 -0
- graph_agents_cli/secrets/cmd_secrets.py +211 -0
- graph_agents_cli/setup/__init__.py +13 -0
- graph_agents_cli/setup/_antigravity.py +221 -0
- graph_agents_cli/setup/cmd_auth.py +1030 -0
- graph_agents_cli/setup/cmd_dev_token.py +513 -0
- graph_agents_cli/setup/cmd_setup.py +428 -0
- graph_agents_cli/setup/cmd_update.py +140 -0
- graph_agents_cli/skills/__init__.py +13 -0
- graph_agents_cli/skills/_bundle.py +65 -0
- graph_agents_cli/skills/data/README.md +19 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/SKILL.md +357 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/references/github-settings.md +113 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/references/gitops.md +137 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/references/kubernetes.md +315 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/references/secrets.md +160 -0
- graph_agents_cli/skills/data/graph-agents-cli-eval/SKILL.md +303 -0
- graph_agents_cli/skills/data/graph-agents-cli-eval/references/dataset_schema.md +282 -0
- graph_agents_cli/skills/data/graph-agents-cli-eval/references/metrics-guide.md +143 -0
- graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/SKILL.md +659 -0
- graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/references/langchain-models.md +124 -0
- graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/references/langgraph.md +235 -0
- graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/references/template-contract.md +477 -0
- graph_agents_cli/skills/data/graph-agents-cli-observability/SKILL.md +231 -0
- graph_agents_cli/skills/data/graph-agents-cli-observability/references/langsmith.md +46 -0
- graph_agents_cli/skills/data/graph-agents-cli-observability/references/otel.md +59 -0
- graph_agents_cli/skills/data/graph-agents-cli-scaffold/SKILL.md +414 -0
- graph_agents_cli/skills/data/graph-agents-cli-scaffold/references/flags.md +134 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/SKILL.md +478 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/brainstorming.md +118 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/commands.md +419 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/extension.md +156 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/internals.md +67 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/spec-template.md +56 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/terminology.md +119 -0
- graph_agents_cli/system/__init__.py +15 -0
- graph_agents_cli/system/_apply.py +519 -0
- graph_agents_cli/system/_checks.py +1023 -0
- graph_agents_cli/system/_deploy.py +215 -0
- graph_agents_cli/system/_model.py +363 -0
- graph_agents_cli/system/_system.py +664 -0
- graph_agents_cli/system/_views.py +208 -0
- graph_agents_cli/system/cmd_system.py +423 -0
- graph_agents_cli-0.3.1.dist-info/METADATA +162 -0
- graph_agents_cli-0.3.1.dist-info/RECORD +291 -0
- graph_agents_cli-0.3.1.dist-info/WHEEL +4 -0
- graph_agents_cli-0.3.1.dist-info/entry_points.txt +2 -0
- graph_agents_cli-0.3.1.dist-info/licenses/LICENSE +201 -0
- graph_agents_cli-0.3.1.dist-info/licenses/NOTICE +19 -0
|
@@ -0,0 +1,420 @@
|
|
|
1
|
+
# Copyright 2026 graph-agents-cli contributors
|
|
2
|
+
#
|
|
3
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
4
|
+
# you may not use this file except in compliance with the License.
|
|
5
|
+
# You may obtain a copy of the License at
|
|
6
|
+
#
|
|
7
|
+
# https://www.apache.org/licenses/LICENSE-2.0
|
|
8
|
+
#
|
|
9
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
10
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
11
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
12
|
+
# See the License for the specific language governing permissions and
|
|
13
|
+
# limitations under the License.
|
|
14
|
+
|
|
15
|
+
"""The evaluation gate: case statuses, quality rates, exit codes.
|
|
16
|
+
|
|
17
|
+
One rule: complete case accounting, every deterministic check and every
|
|
18
|
+
mandatory judge metric must pass. Only judge metrics listed under
|
|
19
|
+
``quality_metrics:`` may miss their per-case threshold at a rate bounded by
|
|
20
|
+
``min_pass_rate``. Statuses: ``passed``, ``failed``, ``quality_below_threshold``,
|
|
21
|
+
``error``, ``missing``. Exit codes: 0 gate met; 1 any failed or a quality
|
|
22
|
+
metric under its rate; 2 any error or missing; 3 configuration error.
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
from __future__ import annotations
|
|
26
|
+
|
|
27
|
+
from dataclasses import dataclass, field
|
|
28
|
+
from typing import Any
|
|
29
|
+
|
|
30
|
+
from rich.markup import escape
|
|
31
|
+
from rich.table import Table
|
|
32
|
+
|
|
33
|
+
from graph_agents_cli._output import Console
|
|
34
|
+
from graph_agents_cli.eval._common import (
|
|
35
|
+
EXIT_GATE_FAILED,
|
|
36
|
+
EXIT_INCOMPLETE,
|
|
37
|
+
EXIT_OK,
|
|
38
|
+
EvalConfigError,
|
|
39
|
+
)
|
|
40
|
+
from graph_agents_cli.eval.checks import run_checks
|
|
41
|
+
from graph_agents_cli.eval.config import EvalConfig, render_judge_prompt, resolve_threshold
|
|
42
|
+
from graph_agents_cli.eval.dataset import SCOPE_ALL_TURNS, EvalCase
|
|
43
|
+
from graph_agents_cli.eval.transcript import RenderedCase, render_case
|
|
44
|
+
|
|
45
|
+
STATUS_PASSED = "passed"
|
|
46
|
+
STATUS_FAILED = "failed"
|
|
47
|
+
STATUS_QUALITY = "quality_below_threshold"
|
|
48
|
+
STATUS_ERROR = "error"
|
|
49
|
+
STATUS_MISSING = "missing"
|
|
50
|
+
STATUSES: tuple[str, ...] = (
|
|
51
|
+
STATUS_PASSED,
|
|
52
|
+
STATUS_FAILED,
|
|
53
|
+
STATUS_QUALITY,
|
|
54
|
+
STATUS_ERROR,
|
|
55
|
+
STATUS_MISSING,
|
|
56
|
+
)
|
|
57
|
+
# Higher is worse; used by ``eval compare`` to detect regressions.
|
|
58
|
+
STATUS_RANK: dict[str, int] = {
|
|
59
|
+
STATUS_PASSED: 0,
|
|
60
|
+
STATUS_QUALITY: 1,
|
|
61
|
+
STATUS_FAILED: 2,
|
|
62
|
+
STATUS_ERROR: 3,
|
|
63
|
+
STATUS_MISSING: 3,
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
_STATUS_STYLE = {
|
|
67
|
+
STATUS_PASSED: "green",
|
|
68
|
+
STATUS_QUALITY: "yellow",
|
|
69
|
+
STATUS_FAILED: "red",
|
|
70
|
+
STATUS_ERROR: "red",
|
|
71
|
+
STATUS_MISSING: "magenta",
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
@dataclass
|
|
76
|
+
class CaseGrade:
|
|
77
|
+
id: str
|
|
78
|
+
checks: dict[str, dict[str, Any]] = field(default_factory=dict)
|
|
79
|
+
judge_scores: dict[str, dict[str, Any]] = field(default_factory=dict)
|
|
80
|
+
missing: list[str] = field(default_factory=list)
|
|
81
|
+
errors: list[str] = field(default_factory=list)
|
|
82
|
+
failures: list[str] = field(default_factory=list)
|
|
83
|
+
quality_misses: list[str] = field(default_factory=list)
|
|
84
|
+
# What the judges of this case could not see in full (cut tool results,
|
|
85
|
+
# unrecorded earlier turns); reported, never silent.
|
|
86
|
+
judge_notes: list[str] = field(default_factory=list)
|
|
87
|
+
truncated_tool_results: int = 0
|
|
88
|
+
unrecorded_turns: int = 0
|
|
89
|
+
# `expect.scope: all_turns` on a multi-turn case whose trace has no per-turn
|
|
90
|
+
# records: its checks could read the final turn only.
|
|
91
|
+
checks_final_turn_only: bool = False
|
|
92
|
+
|
|
93
|
+
@property
|
|
94
|
+
def status(self) -> str:
|
|
95
|
+
if self.missing:
|
|
96
|
+
return STATUS_MISSING
|
|
97
|
+
if self.errors:
|
|
98
|
+
return STATUS_ERROR
|
|
99
|
+
if self.failures:
|
|
100
|
+
return STATUS_FAILED
|
|
101
|
+
if self.quality_misses:
|
|
102
|
+
return STATUS_QUALITY
|
|
103
|
+
return STATUS_PASSED
|
|
104
|
+
|
|
105
|
+
@property
|
|
106
|
+
def reasons(self) -> list[str]:
|
|
107
|
+
return [*self.missing, *self.errors, *self.failures, *self.quality_misses]
|
|
108
|
+
|
|
109
|
+
@property
|
|
110
|
+
def judgeable(self) -> bool:
|
|
111
|
+
"""Judges run only for cases whose mandatory items have not already failed."""
|
|
112
|
+
return not (self.missing or self.errors or self.failures)
|
|
113
|
+
|
|
114
|
+
def to_dict(self) -> dict[str, Any]:
|
|
115
|
+
data: dict[str, Any] = {
|
|
116
|
+
"id": self.id,
|
|
117
|
+
"status": self.status,
|
|
118
|
+
"reasons": self.reasons,
|
|
119
|
+
"checks": self.checks,
|
|
120
|
+
"judge_scores": self.judge_scores,
|
|
121
|
+
}
|
|
122
|
+
if self.judge_notes:
|
|
123
|
+
data["judge_notes"] = list(self.judge_notes)
|
|
124
|
+
return data
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
# --- deterministic stage ------------------------------------------------------
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def grade_deterministic(case: EvalCase, trace: dict[str, Any] | None) -> CaseGrade:
|
|
131
|
+
"""Case accounting plus the deterministic checks the case declares."""
|
|
132
|
+
grade = CaseGrade(id=case.id)
|
|
133
|
+
if trace is None:
|
|
134
|
+
grade.missing.append("missing: no trace for case")
|
|
135
|
+
return grade
|
|
136
|
+
if trace.get("status") == "missing":
|
|
137
|
+
grade.missing.append(f"missing: {trace.get('error') or 'trace has no response'}")
|
|
138
|
+
return grade
|
|
139
|
+
if trace.get("status") == "error":
|
|
140
|
+
grade.errors.append(f"error: {trace.get('error') or 'generation failed'}")
|
|
141
|
+
return grade
|
|
142
|
+
if trace.get("response") is None:
|
|
143
|
+
grade.missing.append("missing: trace has no response")
|
|
144
|
+
return grade
|
|
145
|
+
grade.checks = run_checks(case.expect, trace)
|
|
146
|
+
turns = trace.get("turns")
|
|
147
|
+
grade.checks_final_turn_only = (
|
|
148
|
+
case.expect.get("scope") == SCOPE_ALL_TURNS
|
|
149
|
+
and len(case.user_messages()) > 1
|
|
150
|
+
and not (isinstance(turns, list) and turns)
|
|
151
|
+
)
|
|
152
|
+
for name, result in grade.checks.items():
|
|
153
|
+
if not result["passed"]:
|
|
154
|
+
grade.failures.append(f"{name}: {result['reason']}")
|
|
155
|
+
return grade
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
# --- judge stage --------------------------------------------------------------
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def case_metrics(config: EvalConfig, case: EvalCase) -> dict[str, dict[str, Any]]:
|
|
162
|
+
"""Metric name -> case-level spec for every judge/custom metric that applies."""
|
|
163
|
+
metrics: dict[str, dict[str, Any]] = {name: dict(spec) for name, spec in case.judge.items()}
|
|
164
|
+
for name in config.custom_metrics:
|
|
165
|
+
metrics.setdefault(name, {})
|
|
166
|
+
return metrics
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
def validate_case_metrics(config: EvalConfig, case: EvalCase) -> None:
|
|
170
|
+
"""Unknown metric or missing threshold is a configuration error (exit 3)."""
|
|
171
|
+
known = config.metric_names()
|
|
172
|
+
for name, spec in case_metrics(config, case).items():
|
|
173
|
+
if name not in known:
|
|
174
|
+
raise EvalConfigError(
|
|
175
|
+
f"case {case.id!r} declares unknown judge metric {name!r} "
|
|
176
|
+
f"(known: {', '.join(sorted(known))})"
|
|
177
|
+
)
|
|
178
|
+
resolve_threshold(config, name, spec)
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
def item_id(case_id: str, metric: str) -> str:
|
|
182
|
+
return f"{case_id}/{metric}"
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
def plan_judge_items(
|
|
186
|
+
config: EvalConfig, case: EvalCase, trace: dict[str, Any], grade: CaseGrade
|
|
187
|
+
) -> list[dict[str, Any]]:
|
|
188
|
+
"""Runner items for the judge and custom metrics of one judgeable case."""
|
|
189
|
+
if not grade.judgeable:
|
|
190
|
+
return []
|
|
191
|
+
items: list[dict[str, Any]] = []
|
|
192
|
+
response = trace.get("response")
|
|
193
|
+
response = "" if response is None else str(response)
|
|
194
|
+
rendered: RenderedCase | None = None
|
|
195
|
+
for name, spec in case_metrics(config, case).items():
|
|
196
|
+
threshold = resolve_threshold(config, name, spec)
|
|
197
|
+
quality = config.is_quality(name)
|
|
198
|
+
custom = config.custom_metrics.get(name)
|
|
199
|
+
grade.judge_scores[name] = {
|
|
200
|
+
"score": None,
|
|
201
|
+
"threshold": threshold,
|
|
202
|
+
"passed": None,
|
|
203
|
+
"quality": quality,
|
|
204
|
+
"reasoning": "",
|
|
205
|
+
"error": None,
|
|
206
|
+
# "judge" (an LLM judge) or "custom" (a project callable): errors say which.
|
|
207
|
+
"kind": "custom" if custom is not None else "judge",
|
|
208
|
+
}
|
|
209
|
+
if custom is not None:
|
|
210
|
+
items.append(
|
|
211
|
+
{
|
|
212
|
+
"id": item_id(case.id, name),
|
|
213
|
+
"kind": "custom",
|
|
214
|
+
"case_id": case.id,
|
|
215
|
+
"metric": name,
|
|
216
|
+
"callable": custom.callable,
|
|
217
|
+
"case": case.raw,
|
|
218
|
+
"trace": trace,
|
|
219
|
+
}
|
|
220
|
+
)
|
|
221
|
+
continue
|
|
222
|
+
judge = config.judges[name]
|
|
223
|
+
if rendered is None:
|
|
224
|
+
# Judges see every turn (earlier replies and tool results), not only
|
|
225
|
+
# the user messages and the final reply.
|
|
226
|
+
rendered = render_case(case, trace, max_tool_result_chars=config.max_tool_result_chars)
|
|
227
|
+
grade.judge_notes = rendered.notes
|
|
228
|
+
grade.truncated_tool_results = rendered.truncated_results
|
|
229
|
+
grade.unrecorded_turns = rendered.unrecorded_turns
|
|
230
|
+
items.append(
|
|
231
|
+
{
|
|
232
|
+
"id": item_id(case.id, name),
|
|
233
|
+
"kind": "judge",
|
|
234
|
+
"case_id": case.id,
|
|
235
|
+
"metric": name,
|
|
236
|
+
"scale": judge.scale,
|
|
237
|
+
"prompt": render_judge_prompt(
|
|
238
|
+
judge,
|
|
239
|
+
conversation=rendered.conversation,
|
|
240
|
+
response=response,
|
|
241
|
+
reference=case.reference,
|
|
242
|
+
context=case.context,
|
|
243
|
+
tool_calls=rendered.final_tool_calls or None,
|
|
244
|
+
transcript=rendered.transcript,
|
|
245
|
+
max_tool_result_chars=config.max_tool_result_chars,
|
|
246
|
+
),
|
|
247
|
+
}
|
|
248
|
+
)
|
|
249
|
+
return items
|
|
250
|
+
|
|
251
|
+
|
|
252
|
+
def apply_judge_results(grade: CaseGrade, results: dict[str, dict[str, Any]]) -> None:
|
|
253
|
+
"""Fold runner results into the grade: error, mandatory fail, or quality miss."""
|
|
254
|
+
for name, entry in grade.judge_scores.items():
|
|
255
|
+
label = "custom metric" if entry.get("kind") == "custom" else "judge"
|
|
256
|
+
result = results.get(item_id(grade.id, name))
|
|
257
|
+
if result is None:
|
|
258
|
+
entry["error"] = "no result from judge runner"
|
|
259
|
+
grade.errors.append(f"error: {label} {name}: no result from judge runner")
|
|
260
|
+
continue
|
|
261
|
+
if result.get("error"):
|
|
262
|
+
entry["error"] = str(result["error"])
|
|
263
|
+
grade.errors.append(f"error: {label} {name}: {result['error']}")
|
|
264
|
+
continue
|
|
265
|
+
score = result.get("score")
|
|
266
|
+
if not isinstance(score, int | float) or isinstance(score, bool):
|
|
267
|
+
entry["error"] = f"{label} returned no numeric score ({score!r})"
|
|
268
|
+
grade.errors.append(f"error: {label} {name}: no numeric score")
|
|
269
|
+
continue
|
|
270
|
+
entry["score"] = float(score)
|
|
271
|
+
entry["reasoning"] = str(result.get("reasoning") or "")
|
|
272
|
+
passed = float(score) >= float(entry["threshold"])
|
|
273
|
+
entry["passed"] = passed
|
|
274
|
+
if passed:
|
|
275
|
+
continue
|
|
276
|
+
text = f"{name}: score {_fmt(score)} below threshold {_fmt(entry['threshold'])}"
|
|
277
|
+
if entry["quality"]:
|
|
278
|
+
grade.quality_misses.append(f"{text} (quality)")
|
|
279
|
+
else:
|
|
280
|
+
grade.failures.append(text)
|
|
281
|
+
|
|
282
|
+
|
|
283
|
+
def _fmt(value: Any) -> str:
|
|
284
|
+
if isinstance(value, float) and value.is_integer():
|
|
285
|
+
return str(int(value))
|
|
286
|
+
return str(value)
|
|
287
|
+
|
|
288
|
+
|
|
289
|
+
# --- aggregate ----------------------------------------------------------------
|
|
290
|
+
|
|
291
|
+
|
|
292
|
+
def summarize(grades: list[CaseGrade]) -> dict[str, int]:
|
|
293
|
+
summary = dict.fromkeys(STATUSES, 0)
|
|
294
|
+
for grade in grades:
|
|
295
|
+
summary[grade.status] += 1
|
|
296
|
+
return summary
|
|
297
|
+
|
|
298
|
+
|
|
299
|
+
def compute_quality(config: EvalConfig, grades: list[CaseGrade]) -> dict[str, dict[str, Any]]:
|
|
300
|
+
"""Per quality metric: pass rate over the cases scored on it, and ``met``.
|
|
301
|
+
|
|
302
|
+
The denominator (``scored``) is the cases the metric actually scored: a
|
|
303
|
+
case that never declared the metric is not a pass. ``pass_rate`` and
|
|
304
|
+
``met`` are null on an incomplete run (``status: incomplete``) and when no
|
|
305
|
+
case ran the metric (``status: not_run``, which cannot fail the gate: there
|
|
306
|
+
is nothing to measure, and every mandatory item still applies).
|
|
307
|
+
"""
|
|
308
|
+
complete = bool(grades) and all(g.status not in (STATUS_ERROR, STATUS_MISSING) for g in grades)
|
|
309
|
+
quality: dict[str, dict[str, Any]] = {}
|
|
310
|
+
for name, metric in config.quality_metrics.items():
|
|
311
|
+
scored = [
|
|
312
|
+
entry
|
|
313
|
+
for g in grades
|
|
314
|
+
if (entry := g.judge_scores.get(name)) is not None
|
|
315
|
+
and entry.get("quality")
|
|
316
|
+
and entry.get("passed") is not None
|
|
317
|
+
]
|
|
318
|
+
below = sum(1 for entry in scored if entry.get("passed") is False)
|
|
319
|
+
result: dict[str, Any] = {
|
|
320
|
+
"pass_rate": None,
|
|
321
|
+
"min_pass_rate": metric.min_pass_rate,
|
|
322
|
+
"met": None,
|
|
323
|
+
"below_threshold": below,
|
|
324
|
+
"scored": len(scored),
|
|
325
|
+
"passed": len(scored) - below,
|
|
326
|
+
}
|
|
327
|
+
if not complete:
|
|
328
|
+
result["status"] = "incomplete"
|
|
329
|
+
elif not scored:
|
|
330
|
+
result["status"] = "not_run"
|
|
331
|
+
else:
|
|
332
|
+
rate = (len(scored) - below) / len(scored)
|
|
333
|
+
result["pass_rate"] = round(rate, 4)
|
|
334
|
+
result["met"] = rate >= metric.min_pass_rate
|
|
335
|
+
result["status"] = "met" if result["met"] else "not_met"
|
|
336
|
+
quality[name] = result
|
|
337
|
+
return quality
|
|
338
|
+
|
|
339
|
+
|
|
340
|
+
def exit_code_for(summary: dict[str, int], quality: dict[str, dict[str, Any]]) -> int:
|
|
341
|
+
if summary.get(STATUS_ERROR) or summary.get(STATUS_MISSING):
|
|
342
|
+
return EXIT_INCOMPLETE
|
|
343
|
+
if summary.get(STATUS_FAILED) or any(q.get("met") is False for q in quality.values()):
|
|
344
|
+
return EXIT_GATE_FAILED
|
|
345
|
+
return EXIT_OK
|
|
346
|
+
|
|
347
|
+
|
|
348
|
+
def print_summary(console: Console, results: dict[str, Any]) -> None:
|
|
349
|
+
summary = results["summary"]
|
|
350
|
+
table = Table(title="Evaluation gate", show_header=True, header_style="bold")
|
|
351
|
+
table.add_column("Status")
|
|
352
|
+
table.add_column("Cases", justify="right")
|
|
353
|
+
for status in STATUSES:
|
|
354
|
+
count = summary.get(status, 0)
|
|
355
|
+
style = _STATUS_STYLE[status] if count else "dim"
|
|
356
|
+
table.add_row(f"[{style}]{status}[/{style}]", str(count))
|
|
357
|
+
table.add_row(
|
|
358
|
+
"[bold]planned[/bold]",
|
|
359
|
+
str(results.get("planned", sum(summary.get(s, 0) for s in STATUSES))),
|
|
360
|
+
)
|
|
361
|
+
console.print(table)
|
|
362
|
+
|
|
363
|
+
quality = results.get("quality") or {}
|
|
364
|
+
if quality:
|
|
365
|
+
qtable = Table(title="Quality metrics", show_header=True, header_style="bold")
|
|
366
|
+
qtable.add_column("Metric")
|
|
367
|
+
qtable.add_column("Pass rate", justify="right")
|
|
368
|
+
qtable.add_column("Passed/scored", justify="right")
|
|
369
|
+
qtable.add_column("Min", justify="right")
|
|
370
|
+
qtable.add_column("Met")
|
|
371
|
+
incomplete = bool(summary.get(STATUS_ERROR) or summary.get(STATUS_MISSING))
|
|
372
|
+
for name, entry in quality.items():
|
|
373
|
+
rate = entry.get("pass_rate")
|
|
374
|
+
met = entry.get("met")
|
|
375
|
+
if met is None:
|
|
376
|
+
not_run = entry.get("status") == "not_run" or (
|
|
377
|
+
not incomplete and entry.get("scored") == 0
|
|
378
|
+
)
|
|
379
|
+
met_text = "n/a (no case ran it)" if not_run else "n/a (incomplete)"
|
|
380
|
+
else:
|
|
381
|
+
met_text = "[green]yes[/green]" if met else "[red]no[/red]"
|
|
382
|
+
scored = entry.get("scored")
|
|
383
|
+
qtable.add_row(
|
|
384
|
+
name,
|
|
385
|
+
"n/a" if rate is None else f"{rate:.0%}",
|
|
386
|
+
"n/a" if scored is None else f"{entry.get('passed', 0)}/{scored}",
|
|
387
|
+
f"{entry.get('min_pass_rate', 1.0):.0%}",
|
|
388
|
+
met_text,
|
|
389
|
+
)
|
|
390
|
+
console.print(qtable)
|
|
391
|
+
|
|
392
|
+
flagged = [c for c in results.get("cases", []) if c["status"] != STATUS_PASSED]
|
|
393
|
+
if flagged:
|
|
394
|
+
ctable = Table(title="Cases needing attention", show_header=True, header_style="bold")
|
|
395
|
+
ctable.add_column("Case")
|
|
396
|
+
ctable.add_column("Status")
|
|
397
|
+
ctable.add_column("Reasons")
|
|
398
|
+
for case in flagged:
|
|
399
|
+
style = _STATUS_STYLE[case["status"]]
|
|
400
|
+
ctable.add_row(
|
|
401
|
+
case["id"],
|
|
402
|
+
f"[{style}]{case['status']}[/{style}]",
|
|
403
|
+
"\n".join(case["reasons"][:4]) + ("\n..." if len(case["reasons"]) > 4 else ""),
|
|
404
|
+
)
|
|
405
|
+
console.print(ctable)
|
|
406
|
+
warnings = results.get("warnings") or []
|
|
407
|
+
for text in warnings:
|
|
408
|
+
console.print(f"[bold yellow]Warning:[/bold yellow] [yellow]{escape(text)}[/yellow]")
|
|
409
|
+
code = summary.get("exit_code", 0)
|
|
410
|
+
verdict = {
|
|
411
|
+
0: "[green]gate met[/green]",
|
|
412
|
+
1: "[red]gate failed[/red]",
|
|
413
|
+
2: "[magenta]incomplete run[/magenta]",
|
|
414
|
+
}
|
|
415
|
+
caveat = ""
|
|
416
|
+
if code == 0 and results.get("fake_model"):
|
|
417
|
+
caveat = (
|
|
418
|
+
" [bold yellow](fake model: plumbing check only, not a quality signal)[/bold yellow]"
|
|
419
|
+
)
|
|
420
|
+
console.print(f"Result: {verdict.get(code, code)} (exit code {code}){caveat}")
|
|
@@ -0,0 +1,192 @@
|
|
|
1
|
+
# Copyright 2026 graph-agents-cli contributors
|
|
2
|
+
#
|
|
3
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
4
|
+
# you may not use this file except in compliance with the License.
|
|
5
|
+
# You may obtain a copy of the License at
|
|
6
|
+
#
|
|
7
|
+
# https://www.apache.org/licenses/LICENSE-2.0
|
|
8
|
+
#
|
|
9
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
10
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
11
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
12
|
+
# See the License for the specific language governing permissions and
|
|
13
|
+
# limitations under the License.
|
|
14
|
+
|
|
15
|
+
"""What a judge sees of one case: the conversation, tool calls and replies of every turn.
|
|
16
|
+
|
|
17
|
+
A multi-turn case sends each user message on one thread, and the trace keeps
|
|
18
|
+
every turn under ``turns`` (the top-level ``response`` and ``tool_calls`` are
|
|
19
|
+
the final turn's). A judge scoring the final reply needs the earlier replies
|
|
20
|
+
and tool results: a follow-up that relies on an answer already given, or a
|
|
21
|
+
claim grounded in an earlier tool result, would otherwise look wrong.
|
|
22
|
+
|
|
23
|
+
Rendered pieces:
|
|
24
|
+
|
|
25
|
+
* ``conversation``: every turn before the reply being scored in full (user
|
|
26
|
+
message, each tool call with its result, the agent's reply), then the final
|
|
27
|
+
user message. Dataset ``system``/``assistant`` messages, which are not sent
|
|
28
|
+
to the agent, appear in place and are labelled as such.
|
|
29
|
+
* ``transcript``: the same plus the final turn's tool calls and reply.
|
|
30
|
+
* the final turn's tool calls, rendered by ``render_tool_calls``.
|
|
31
|
+
|
|
32
|
+
Tool results longer than the configured limit are cut with an explicit marker
|
|
33
|
+
that tells the judge how much it did not see, so a claim drawn from the omitted
|
|
34
|
+
part is never mistaken for an invented one.
|
|
35
|
+
"""
|
|
36
|
+
|
|
37
|
+
from __future__ import annotations
|
|
38
|
+
|
|
39
|
+
import json
|
|
40
|
+
from dataclasses import dataclass, field
|
|
41
|
+
from typing import Any
|
|
42
|
+
|
|
43
|
+
from graph_agents_cli.eval.dataset import EvalCase
|
|
44
|
+
|
|
45
|
+
# Characters of one tool result shown to a judge before it is cut. The agent
|
|
46
|
+
# read the whole result, so the judge needs it too; the bound only keeps one
|
|
47
|
+
# runaway tool from exhausting the judge's context. Configurable per project as
|
|
48
|
+
# ``judge.max_tool_result_chars`` in eval_config.yaml (null = never cut).
|
|
49
|
+
DEFAULT_MAX_TOOL_RESULT_CHARS = 50_000
|
|
50
|
+
|
|
51
|
+
NOT_SENT = "(from the dataset; not sent to the agent)"
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
@dataclass
|
|
55
|
+
class RenderedCase:
|
|
56
|
+
conversation: str
|
|
57
|
+
transcript: str
|
|
58
|
+
final_tool_calls: list[dict[str, Any]] = field(default_factory=list)
|
|
59
|
+
# Tool results (in any turn) cut for the judge, and how many characters went unseen.
|
|
60
|
+
truncated_results: int = 0
|
|
61
|
+
truncated_chars: int = 0
|
|
62
|
+
# Earlier turns whose replies the trace does not record (an older trace or a
|
|
63
|
+
# generate override): the judge is told, rather than shown a gap silently.
|
|
64
|
+
unrecorded_turns: int = 0
|
|
65
|
+
|
|
66
|
+
@property
|
|
67
|
+
def notes(self) -> list[str]:
|
|
68
|
+
notes: list[str] = []
|
|
69
|
+
if self.truncated_results:
|
|
70
|
+
notes.append(
|
|
71
|
+
f"{self.truncated_results} tool result(s) cut for the judge "
|
|
72
|
+
f"({self.truncated_chars} characters not shown; raise "
|
|
73
|
+
"judge.max_tool_result_chars in eval_config.yaml to show more)"
|
|
74
|
+
)
|
|
75
|
+
if self.unrecorded_turns:
|
|
76
|
+
notes.append(
|
|
77
|
+
f"{self.unrecorded_turns} earlier turn(s) have no recorded reply in the trace "
|
|
78
|
+
"(re-run eval generate to record every turn)"
|
|
79
|
+
)
|
|
80
|
+
return notes
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def _as_text(value: Any) -> str:
|
|
84
|
+
if value is None:
|
|
85
|
+
return ""
|
|
86
|
+
if isinstance(value, str):
|
|
87
|
+
return value
|
|
88
|
+
try:
|
|
89
|
+
return json.dumps(value, ensure_ascii=False, sort_keys=True, default=str)
|
|
90
|
+
except (TypeError, ValueError):
|
|
91
|
+
return str(value)
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _args_text(args: Any) -> str:
|
|
95
|
+
if args is None or args == {}:
|
|
96
|
+
return ""
|
|
97
|
+
return _as_text(args)
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def cut_tool_result(text: str, max_chars: int | None) -> tuple[str, int]:
|
|
101
|
+
"""``(text shown to the judge, characters cut)``; the cut is marked in-band."""
|
|
102
|
+
if max_chars is None or len(text) <= max_chars:
|
|
103
|
+
return text, 0
|
|
104
|
+
omitted = len(text) - max_chars
|
|
105
|
+
marker = (
|
|
106
|
+
f" [TRUNCATED by graph-agents-cli: the judge sees the first {max_chars} of "
|
|
107
|
+
f"{len(text)} characters of this tool result; {omitted} more were returned to the "
|
|
108
|
+
"agent but are not shown here. Do not treat a claim as unsupported only because "
|
|
109
|
+
"it could come from the omitted part.]"
|
|
110
|
+
)
|
|
111
|
+
return text[:max_chars] + marker, omitted
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def render_tool_call(call: dict[str, Any], max_chars: int | None) -> tuple[str, int]:
|
|
115
|
+
"""``name(args) -> result`` for one call, and how many result characters were cut."""
|
|
116
|
+
result, omitted = cut_tool_result(_as_text(call.get("result")), max_chars)
|
|
117
|
+
flag = " (error)" if call.get("is_error") else ""
|
|
118
|
+
return f"{call.get('name')}({_args_text(call.get('args'))}) -> {result}{flag}", omitted
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def render_tool_calls(
|
|
122
|
+
calls: list[dict[str, Any]] | None, max_chars: int | None
|
|
123
|
+
) -> tuple[list[str], int, int]:
|
|
124
|
+
"""Rendered lines, the number of results cut, and the characters cut."""
|
|
125
|
+
lines: list[str] = []
|
|
126
|
+
cut = 0
|
|
127
|
+
omitted_total = 0
|
|
128
|
+
for call in calls or []:
|
|
129
|
+
if not isinstance(call, dict):
|
|
130
|
+
continue
|
|
131
|
+
line, omitted = render_tool_call(call, max_chars)
|
|
132
|
+
lines.append(line)
|
|
133
|
+
if omitted:
|
|
134
|
+
cut += 1
|
|
135
|
+
omitted_total += omitted
|
|
136
|
+
return lines, cut, omitted_total
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def render_case(
|
|
140
|
+
case: EvalCase,
|
|
141
|
+
trace: dict[str, Any],
|
|
142
|
+
*,
|
|
143
|
+
max_tool_result_chars: int | None = DEFAULT_MAX_TOOL_RESULT_CHARS,
|
|
144
|
+
) -> RenderedCase:
|
|
145
|
+
"""Render the conversation and full transcript of one graded case."""
|
|
146
|
+
user_count = sum(1 for m in case.messages if m.get("role") == "user")
|
|
147
|
+
raw_turns = trace.get("turns")
|
|
148
|
+
turns: list[dict[str, Any]] = (
|
|
149
|
+
[t if isinstance(t, dict) else {} for t in raw_turns] if isinstance(raw_turns, list) else []
|
|
150
|
+
)
|
|
151
|
+
final_calls = [c for c in (trace.get("tool_calls") or []) if isinstance(c, dict)]
|
|
152
|
+
final_response = trace.get("response")
|
|
153
|
+
labelled = user_count > 1
|
|
154
|
+
|
|
155
|
+
before: list[str] = [] # everything up to and including the final user message
|
|
156
|
+
after: list[str] = [] # the final turn's tool calls and reply (transcript only)
|
|
157
|
+
rendered = RenderedCase(conversation="", transcript="", final_tool_calls=final_calls)
|
|
158
|
+
|
|
159
|
+
def _calls(lines_out: list[str], calls: list[dict[str, Any]]) -> None:
|
|
160
|
+
lines, cut, omitted = render_tool_calls(calls, max_tool_result_chars)
|
|
161
|
+
rendered.truncated_results += cut
|
|
162
|
+
rendered.truncated_chars += omitted
|
|
163
|
+
lines_out.extend(f"agent tool call: {line}" for line in lines)
|
|
164
|
+
|
|
165
|
+
turn_index = -1
|
|
166
|
+
for message in case.messages:
|
|
167
|
+
role = message.get("role", "?")
|
|
168
|
+
content = str(message.get("content", ""))
|
|
169
|
+
if role != "user":
|
|
170
|
+
# Dataset context for the judge; the agent never received it.
|
|
171
|
+
before.append(f"{role} {NOT_SENT}: {content}")
|
|
172
|
+
continue
|
|
173
|
+
turn_index += 1
|
|
174
|
+
if labelled:
|
|
175
|
+
before.append(f"--- turn {turn_index + 1} of {user_count} ---")
|
|
176
|
+
before.append(f"user: {content}")
|
|
177
|
+
if turn_index == user_count - 1:
|
|
178
|
+
continue # the final turn's calls and reply are the ones being scored
|
|
179
|
+
record = turns[turn_index] if turn_index < len(turns) else None
|
|
180
|
+
if record is None:
|
|
181
|
+
rendered.unrecorded_turns += 1
|
|
182
|
+
before.append("agent: [reply not recorded in the trace]")
|
|
183
|
+
continue
|
|
184
|
+
_calls(before, [c for c in (record.get("tool_calls") or []) if isinstance(c, dict)])
|
|
185
|
+
before.append(f"agent: {_as_text(record.get('response'))}")
|
|
186
|
+
|
|
187
|
+
_calls(after, final_calls)
|
|
188
|
+
after.append(f"agent: {_as_text(final_response)}")
|
|
189
|
+
|
|
190
|
+
rendered.conversation = "\n".join(before)
|
|
191
|
+
rendered.transcript = "\n".join(before + after)
|
|
192
|
+
return rendered
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
# Copyright 2026 Google LLC
|
|
2
|
+
#
|
|
3
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
4
|
+
# you may not use this file except in compliance with the License.
|
|
5
|
+
# You may obtain a copy of the License at
|
|
6
|
+
#
|
|
7
|
+
# https://www.apache.org/licenses/LICENSE-2.0
|
|
8
|
+
#
|
|
9
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
10
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
11
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
12
|
+
# See the License for the specific language governing permissions and
|
|
13
|
+
# limitations under the License.
|