graph-agents-cli 0.3.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- graph_agents_cli/__init__.py +26 -0
- graph_agents_cli/_api_policy.py +2145 -0
- graph_agents_cli/_approvals.py +400 -0
- graph_agents_cli/_build.py +186 -0
- graph_agents_cli/_build_info.json +7 -0
- graph_agents_cli/_chat_client.py +462 -0
- graph_agents_cli/_click.py +157 -0
- graph_agents_cli/_defaults.py +139 -0
- graph_agents_cli/_experiments.py +64 -0
- graph_agents_cli/_http.py +192 -0
- graph_agents_cli/_output.py +83 -0
- graph_agents_cli/_project.py +462 -0
- graph_agents_cli/_remote.py +220 -0
- graph_agents_cli/_response_schema.py +264 -0
- graph_agents_cli/_runner.py +319 -0
- graph_agents_cli/_skills_check.py +274 -0
- graph_agents_cli/_tools.py +189 -0
- graph_agents_cli/_trust.py +66 -0
- graph_agents_cli/api/__init__.py +15 -0
- graph_agents_cli/api/_changes.py +506 -0
- graph_agents_cli/api/_files.py +658 -0
- graph_agents_cli/api/cmd_api.py +2480 -0
- graph_agents_cli/deploy/__init__.py +15 -0
- graph_agents_cli/deploy/_config.py +171 -0
- graph_agents_cli/deploy/_image.py +128 -0
- graph_agents_cli/deploy/_kube.py +286 -0
- graph_agents_cli/deploy/_modes.py +234 -0
- graph_agents_cli/deploy/_preflight.py +370 -0
- graph_agents_cli/deploy/_values.py +168 -0
- graph_agents_cli/deploy/cmd_deploy.py +1866 -0
- graph_agents_cli/deploy/gitops.py +562 -0
- graph_agents_cli/deploy/local_load.py +273 -0
- graph_agents_cli/dev/__init__.py +13 -0
- graph_agents_cli/dev/cmd_build.py +131 -0
- graph_agents_cli/dev/cmd_install.py +78 -0
- graph_agents_cli/dev/cmd_lint.py +119 -0
- graph_agents_cli/dev/cmd_playground.py +297 -0
- graph_agents_cli/dev/policy_check.py +1287 -0
- graph_agents_cli/eval/__init__.py +22 -0
- graph_agents_cli/eval/_client.py +670 -0
- graph_agents_cli/eval/_common.py +177 -0
- graph_agents_cli/eval/_judge.py +168 -0
- graph_agents_cli/eval/_judge_runner.py +238 -0
- graph_agents_cli/eval/_paths.py +212 -0
- graph_agents_cli/eval/checks.py +581 -0
- graph_agents_cli/eval/cmd_analyze.py +278 -0
- graph_agents_cli/eval/cmd_compare.py +284 -0
- graph_agents_cli/eval/cmd_eval_group.py +80 -0
- graph_agents_cli/eval/cmd_generate.py +558 -0
- graph_agents_cli/eval/cmd_grade.py +466 -0
- graph_agents_cli/eval/cmd_metric.py +156 -0
- graph_agents_cli/eval/cmd_run.py +370 -0
- graph_agents_cli/eval/cmd_submit.py +400 -0
- graph_agents_cli/eval/config.py +435 -0
- graph_agents_cli/eval/dataset.py +350 -0
- graph_agents_cli/eval/gate.py +420 -0
- graph_agents_cli/eval/transcript.py +192 -0
- graph_agents_cli/extension/__init__.py +13 -0
- graph_agents_cli/extension/_compat.py +86 -0
- graph_agents_cli/extension/_loader.py +293 -0
- graph_agents_cli/extension/_manifest.py +135 -0
- graph_agents_cli/extension/_overrides.py +195 -0
- graph_agents_cli/extension/_paths.py +91 -0
- graph_agents_cli/extension/_refs.py +193 -0
- graph_agents_cli/extension/_resolver.py +453 -0
- graph_agents_cli/extension/_schema.py +106 -0
- graph_agents_cli/extension/_spec.py +253 -0
- graph_agents_cli/extension/_sync.py +102 -0
- graph_agents_cli/extension/_trust.py +58 -0
- graph_agents_cli/extension/cmd_extension_add.py +259 -0
- graph_agents_cli/extension/cmd_extension_group.py +57 -0
- graph_agents_cli/extension/cmd_extension_list.py +56 -0
- graph_agents_cli/extension/cmd_extension_remove.py +61 -0
- graph_agents_cli/extension/cmd_extension_update.py +195 -0
- graph_agents_cli/info/__init__.py +13 -0
- graph_agents_cli/info/cmd_info.py +222 -0
- graph_agents_cli/infra/__init__.py +15 -0
- graph_agents_cli/infra/checks.py +1169 -0
- graph_agents_cli/infra/cmd_infra.py +103 -0
- graph_agents_cli/main.py +591 -0
- graph_agents_cli/peer/__init__.py +15 -0
- graph_agents_cli/peer/_generate.py +254 -0
- graph_agents_cli/peer/cmd_peer.py +1151 -0
- graph_agents_cli/run/__init__.py +13 -0
- graph_agents_cli/run/_local_server.py +1157 -0
- graph_agents_cli/run/_signals.py +141 -0
- graph_agents_cli/run/cmd_approvals.py +530 -0
- graph_agents_cli/run/cmd_run.py +1421 -0
- graph_agents_cli/scaffold/__init__.py +19 -0
- graph_agents_cli/scaffold/agents/README.md +24 -0
- graph_agents_cli/scaffold/agents/empty_py/.template/templateconfig.yaml +22 -0
- graph_agents_cli/scaffold/agents/langgraph/.env.example +292 -0
- graph_agents_cli/scaffold/agents/langgraph/.template/templateconfig.yaml +28 -0
- graph_agents_cli/scaffold/agents/langgraph/Dockerfile +59 -0
- graph_agents_cli/scaffold/agents/langgraph/Dockerfile.langgraph-server +59 -0
- graph_agents_cli/scaffold/agents/langgraph/README.md +571 -0
- graph_agents_cli/scaffold/agents/langgraph/api-policy.yaml +60 -0
- graph_agents_cli/scaffold/agents/langgraph/app/__init__.py +20 -0
- graph_agents_cli/scaffold/agents/langgraph/app/agent.py +174 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/__init__.py +15 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/a2a.py +2162 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/a2a_client.py +1167 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/api_client.py +4220 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/approvals.py +1349 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/auth.py +1986 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/chat.py +2962 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/checkpointer.py +432 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/content.py +569 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/db.py +580 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/limits.py +203 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/metrics.py +231 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/middleware.py +361 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/model.py +611 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/playground.py +230 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/run_locks.py +459 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/structured.py +755 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/telemetry.py +681 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/threads.py +493 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/token_exchange.py +959 -0
- graph_agents_cli/scaffold/agents/langgraph/app/fast_api_app.py +770 -0
- graph_agents_cli/scaffold/agents/langgraph/app/policies/__init__.py +55 -0
- graph_agents_cli/scaffold/agents/langgraph/app/policies/custom.py +97 -0
- graph_agents_cli/scaffold/agents/langgraph/app/tools/__init__.py +46 -0
- graph_agents_cli/scaffold/agents/langgraph/app/tools/example_api.py +92 -0
- graph_agents_cli/scaffold/agents/langgraph/app/tools/weather.py +33 -0
- graph_agents_cli/scaffold/agents/langgraph/langgraph.json +14 -0
- graph_agents_cli/scaffold/agents/langgraph/pyproject.toml +78 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/conftest.py +376 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/eval/datasets/basic-dataset.json +53 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/eval/eval_config.yaml +32 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/approval_graph.py +137 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/fake_issuer.py +216 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/fake_openai.py +357 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_a2a_outcomes.py +569 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_a2a_relay.py +479 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_api_surface.py +812 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_approvals.py +1367 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_approvals_server.py +794 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_cross_actor.py +497 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_cross_actor_server.py +247 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_history_repair.py +278 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_model_apis.py +242 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_postgres.py +637 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_resilience_postgres.py +770 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_runtime_guardrails.py +854 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_server_e2e.py +340 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_server_runtime.py +989 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_structured_answers.py +584 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_structured_server.py +222 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_token_exchange_issuer.py +650 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/load_test/.results/.placeholder +0 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/load_test/README.md +22 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/load_test/conftest.py +21 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/load_test/load_test.py +81 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_a2a_client.py +824 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_a2a_scoping.py +724 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_api_client.py +1214 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_api_client_hardening.py +716 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_api_policy_rpc.py +767 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_approval_ledger.py +1536 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_fake_model.py +115 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_jwt_policy.py +991 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_limits.py +310 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_logging.py +148 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_logging_hardening.py +271 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_policy.py +378 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_resilience.py +610 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_server_auth.py +702 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_structured.py +673 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_telemetry.py +404 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_thread_listing.py +255 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_threads.py +268 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_token_exchange.py +1320 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_untrusted_content.py +393 -0
- graph_agents_cli/scaffold/agents/langgraph/uv-fastapi.lock +2084 -0
- graph_agents_cli/scaffold/agents/langgraph/uv-langgraph-server.lock +2106 -0
- graph_agents_cli/scaffold/agents/langgraph/{{cookiecutter.agent_guidance_filename}} +129 -0
- graph_agents_cli/scaffold/base_templates/_shared/graph-agents-cli-manifest.yaml +36 -0
- graph_agents_cli/scaffold/base_templates/python/.dockerignore +32 -0
- graph_agents_cli/scaffold/base_templates/python/.github/CODEOWNERS +30 -0
- graph_agents_cli/scaffold/base_templates/python/.github/agent.env +7 -0
- graph_agents_cli/scaffold/base_templates/python/.github/workflows/pr_checks.yaml +214 -0
- graph_agents_cli/scaffold/base_templates/python/.gitignore +209 -0
- graph_agents_cli/scaffold/base_templates/python/tests/unit/test_dummy.py +23 -0
- graph_agents_cli/scaffold/base_templates/python/{{cookiecutter.agent_guidance_filename}} +35 -0
- graph_agents_cli/scaffold/cmd_scaffold_group.py +49 -0
- graph_agents_cli/scaffold/commands/__init__.py +13 -0
- graph_agents_cli/scaffold/commands/create.py +1424 -0
- graph_agents_cli/scaffold/commands/enhance.py +1652 -0
- graph_agents_cli/scaffold/commands/upgrade.py +570 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/.github/agent.env +12 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/.github/workflows/promote-to-prod.yaml +371 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/.github/workflows/staging.yaml +450 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/argocd/application-dev.yaml +43 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/argocd/application-prod.yaml +41 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/argocd/application-staging.yaml +43 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/.helmignore +14 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/Chart.yaml +21 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/examples/networkpolicy.yaml +103 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/NOTES.txt +48 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/_helpers.tpl +189 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/certificate.yaml +15 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/configmap.yaml +10 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/deployment.yaml +199 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/hpa.yaml +22 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/httproute.yaml +30 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/ingress.yaml +39 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/networkpolicy.yaml +48 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/pdb.yaml +13 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/postgresql-secret.yaml +37 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/service.yaml +15 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/serviceaccount.yaml +13 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/servicemonitor.yaml +42 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values-dev.yaml +22 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values-prod.yaml +45 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values-staging.yaml +29 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values.yaml +396 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/tests/integration/test_chart.py +269 -0
- graph_agents_cli/scaffold/deployment_targets/none/README.md +5 -0
- graph_agents_cli/scaffold/deployment_targets/none/python/README.md +6 -0
- graph_agents_cli/scaffold/utils/__init__.py +13 -0
- graph_agents_cli/scaffold/utils/backup.py +212 -0
- graph_agents_cli/scaffold/utils/build_record.py +257 -0
- graph_agents_cli/scaffold/utils/cli_options.py +184 -0
- graph_agents_cli/scaffold/utils/fs.py +83 -0
- graph_agents_cli/scaffold/utils/generate_locks.py +214 -0
- graph_agents_cli/scaffold/utils/generation_metadata.py +88 -0
- graph_agents_cli/scaffold/utils/keyedit.py +768 -0
- graph_agents_cli/scaffold/utils/keymerge.py +537 -0
- graph_agents_cli/scaffold/utils/language.py +138 -0
- graph_agents_cli/scaffold/utils/lock_utils.py +94 -0
- graph_agents_cli/scaffold/utils/logging.py +77 -0
- graph_agents_cli/scaffold/utils/manifest.py +292 -0
- graph_agents_cli/scaffold/utils/merge.py +970 -0
- graph_agents_cli/scaffold/utils/merge3.py +216 -0
- graph_agents_cli/scaffold/utils/openapi_seed.py +199 -0
- graph_agents_cli/scaffold/utils/remote_template.py +376 -0
- graph_agents_cli/scaffold/utils/template.py +1352 -0
- graph_agents_cli/scaffold/utils/upgrade.py +894 -0
- graph_agents_cli/scaffold/utils/version.py +438 -0
- graph_agents_cli/secrets/__init__.py +15 -0
- graph_agents_cli/secrets/_apply.py +954 -0
- graph_agents_cli/secrets/_required.py +188 -0
- graph_agents_cli/secrets/cmd_secrets.py +211 -0
- graph_agents_cli/setup/__init__.py +13 -0
- graph_agents_cli/setup/_antigravity.py +221 -0
- graph_agents_cli/setup/cmd_auth.py +1030 -0
- graph_agents_cli/setup/cmd_dev_token.py +513 -0
- graph_agents_cli/setup/cmd_setup.py +428 -0
- graph_agents_cli/setup/cmd_update.py +140 -0
- graph_agents_cli/skills/__init__.py +13 -0
- graph_agents_cli/skills/_bundle.py +65 -0
- graph_agents_cli/skills/data/README.md +19 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/SKILL.md +357 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/references/github-settings.md +113 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/references/gitops.md +137 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/references/kubernetes.md +315 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/references/secrets.md +160 -0
- graph_agents_cli/skills/data/graph-agents-cli-eval/SKILL.md +303 -0
- graph_agents_cli/skills/data/graph-agents-cli-eval/references/dataset_schema.md +282 -0
- graph_agents_cli/skills/data/graph-agents-cli-eval/references/metrics-guide.md +143 -0
- graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/SKILL.md +659 -0
- graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/references/langchain-models.md +124 -0
- graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/references/langgraph.md +235 -0
- graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/references/template-contract.md +477 -0
- graph_agents_cli/skills/data/graph-agents-cli-observability/SKILL.md +231 -0
- graph_agents_cli/skills/data/graph-agents-cli-observability/references/langsmith.md +46 -0
- graph_agents_cli/skills/data/graph-agents-cli-observability/references/otel.md +59 -0
- graph_agents_cli/skills/data/graph-agents-cli-scaffold/SKILL.md +414 -0
- graph_agents_cli/skills/data/graph-agents-cli-scaffold/references/flags.md +134 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/SKILL.md +478 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/brainstorming.md +118 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/commands.md +419 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/extension.md +156 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/internals.md +67 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/spec-template.md +56 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/terminology.md +119 -0
- graph_agents_cli/system/__init__.py +15 -0
- graph_agents_cli/system/_apply.py +519 -0
- graph_agents_cli/system/_checks.py +1023 -0
- graph_agents_cli/system/_deploy.py +215 -0
- graph_agents_cli/system/_model.py +363 -0
- graph_agents_cli/system/_system.py +664 -0
- graph_agents_cli/system/_views.py +208 -0
- graph_agents_cli/system/cmd_system.py +423 -0
- graph_agents_cli-0.3.1.dist-info/METADATA +162 -0
- graph_agents_cli-0.3.1.dist-info/RECORD +291 -0
- graph_agents_cli-0.3.1.dist-info/WHEEL +4 -0
- graph_agents_cli-0.3.1.dist-info/entry_points.txt +2 -0
- graph_agents_cli-0.3.1.dist-info/licenses/LICENSE +201 -0
- graph_agents_cli-0.3.1.dist-info/licenses/NOTICE +19 -0
|
@@ -0,0 +1,581 @@
|
|
|
1
|
+
# Copyright 2026 graph-agents-cli contributors
|
|
2
|
+
#
|
|
3
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
4
|
+
# you may not use this file except in compliance with the License.
|
|
5
|
+
# You may obtain a copy of the License at
|
|
6
|
+
#
|
|
7
|
+
# https://www.apache.org/licenses/LICENSE-2.0
|
|
8
|
+
#
|
|
9
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
10
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
11
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
12
|
+
# See the License for the specific language governing permissions and
|
|
13
|
+
# limitations under the License.
|
|
14
|
+
|
|
15
|
+
"""Deterministic checks: the ``expect`` block of a case, evaluated in the CLI process.
|
|
16
|
+
|
|
17
|
+
Every check is a pure function returning ``(passed, reason)``; ``reason`` is
|
|
18
|
+
a short human sentence (empty when passed). ``run_checks`` applies the checks a
|
|
19
|
+
case declares in ``expect`` to one trace. No model, no network.
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
from __future__ import annotations
|
|
23
|
+
|
|
24
|
+
import json
|
|
25
|
+
import re
|
|
26
|
+
from collections.abc import Callable
|
|
27
|
+
from typing import Any
|
|
28
|
+
|
|
29
|
+
from graph_agents_cli._approvals import call_matches, describe_match
|
|
30
|
+
from graph_agents_cli.eval.dataset import SCOPE_ALL_TURNS
|
|
31
|
+
|
|
32
|
+
CHECK_DESCRIPTIONS: dict[str, str] = {
|
|
33
|
+
"contains": (
|
|
34
|
+
"Every listed substring appears in the response (case-insensitive; "
|
|
35
|
+
"`case_insensitive: false` for exact case)."
|
|
36
|
+
),
|
|
37
|
+
"not_contains": (
|
|
38
|
+
"None of the listed substrings appears in the response (case-insensitive; "
|
|
39
|
+
"`case_insensitive: false` for exact case)."
|
|
40
|
+
),
|
|
41
|
+
"regex": (
|
|
42
|
+
"The response matches the regular expression (re.search, DOTALL; `(?i)` ignores case)."
|
|
43
|
+
),
|
|
44
|
+
"json_schema": (
|
|
45
|
+
"The final answer validates against the schema: the run's structured response when it "
|
|
46
|
+
"has one (a project with a response schema), else the final reply's JSON (the whole "
|
|
47
|
+
"reply, else its last JSON object or array of the schema's type)."
|
|
48
|
+
),
|
|
49
|
+
"tool_calls": "The listed tools were called (name + args_subset); `ordered` enforces order.",
|
|
50
|
+
"no_tool_calls": "The agent made no tool calls.",
|
|
51
|
+
"max_latency_ms": (
|
|
52
|
+
"message.end latency_ms is at or below the limit (every turn's, with scope all_turns)."
|
|
53
|
+
),
|
|
54
|
+
"max_tokens": (
|
|
55
|
+
"usage.input_tokens + usage.output_tokens is at or below the limit "
|
|
56
|
+
"(summed over the turns with scope all_turns)."
|
|
57
|
+
),
|
|
58
|
+
"approvals": (
|
|
59
|
+
"The listed calls (match: operation_id and/or method + path, optional api) hit an "
|
|
60
|
+
"approval gate, each with its status: gated (any outcome), approved or rejected."
|
|
61
|
+
),
|
|
62
|
+
"no_approvals": "No call hit an approval gate.",
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
# Keys of `expect` that change how checks read the trace; they are not checks.
|
|
66
|
+
MODIFIER_DESCRIPTIONS: dict[str, str] = {
|
|
67
|
+
"ordered": "tool_calls must appear in the listed order (default false).",
|
|
68
|
+
"case_insensitive": "contains / not_contains ignore case (default true).",
|
|
69
|
+
"scope": (
|
|
70
|
+
"final_turn (default): checks read the final turn of a multi-turn case; all_turns: "
|
|
71
|
+
"every turn's replies, tool calls, latency, tokens and approval gates (json_schema "
|
|
72
|
+
"always reads the final answer)."
|
|
73
|
+
),
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
CheckResult = tuple[bool, str]
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def _short(text: str, limit: int = 80) -> str:
|
|
80
|
+
text = text.replace("\n", " ")
|
|
81
|
+
return text if len(text) <= limit else text[: limit - 3] + "..."
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def _replies(response: str | list[str]) -> list[str]:
|
|
85
|
+
return [response] if isinstance(response, str) else list(response)
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def _occurs(needle: str, haystacks: list[str], case_insensitive: bool) -> bool:
|
|
89
|
+
if case_insensitive:
|
|
90
|
+
folded = needle.casefold()
|
|
91
|
+
return any(folded in h.casefold() for h in haystacks)
|
|
92
|
+
return any(needle in h for h in haystacks)
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def check_contains(
|
|
96
|
+
expected: list[str], response: str | list[str], *, case_insensitive: bool = True
|
|
97
|
+
) -> CheckResult:
|
|
98
|
+
"""Every substring occurs in the response (or, given every turn's replies, in one of them)."""
|
|
99
|
+
replies = _replies(response)
|
|
100
|
+
missing = [s for s in expected if not _occurs(s, replies, case_insensitive)]
|
|
101
|
+
if missing:
|
|
102
|
+
subject = "response does not" if isinstance(response, str) else "no reply in any turn"
|
|
103
|
+
verb = " contain " if isinstance(response, str) else " contains "
|
|
104
|
+
mode = "" if case_insensitive else " (exact case)"
|
|
105
|
+
return False, f"{subject}{verb}" + ", ".join(repr(s) for s in missing) + mode
|
|
106
|
+
return True, ""
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def check_not_contains(
|
|
110
|
+
forbidden: list[str], response: str | list[str], *, case_insensitive: bool = True
|
|
111
|
+
) -> CheckResult:
|
|
112
|
+
"""No substring occurs in the response (or in any turn's reply)."""
|
|
113
|
+
replies = _replies(response)
|
|
114
|
+
present = [s for s in forbidden if _occurs(s, replies, case_insensitive)]
|
|
115
|
+
if present:
|
|
116
|
+
subject = "response contains" if isinstance(response, str) else "a turn's reply contains"
|
|
117
|
+
mode = " (ignoring case)" if case_insensitive else ""
|
|
118
|
+
return False, f"{subject} forbidden " + ", ".join(repr(s) for s in present) + mode
|
|
119
|
+
return True, ""
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def check_regex(pattern: str, response: str | list[str]) -> CheckResult:
|
|
123
|
+
"""``re.search`` matches the response (or at least one turn's reply)."""
|
|
124
|
+
try:
|
|
125
|
+
compiled = re.compile(pattern, re.DOTALL)
|
|
126
|
+
except re.error as exc:
|
|
127
|
+
return False, f"invalid regex {pattern!r}: {exc}"
|
|
128
|
+
if not any(compiled.search(text) is not None for text in _replies(response)):
|
|
129
|
+
subject = "response does not" if isinstance(response, str) else "no reply in any turn"
|
|
130
|
+
verb = " match " if isinstance(response, str) else " matches "
|
|
131
|
+
return False, f"{subject}{verb}/{pattern}/"
|
|
132
|
+
return True, ""
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
# --- json_schema -------------------------------------------------------------
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def _type_matches(value: Any, type_name: str) -> bool:
|
|
139
|
+
match type_name:
|
|
140
|
+
case "object":
|
|
141
|
+
return isinstance(value, dict)
|
|
142
|
+
case "array":
|
|
143
|
+
return isinstance(value, list)
|
|
144
|
+
case "string":
|
|
145
|
+
return isinstance(value, str)
|
|
146
|
+
case "integer":
|
|
147
|
+
return isinstance(value, int) and not isinstance(value, bool)
|
|
148
|
+
case "number":
|
|
149
|
+
return isinstance(value, int | float) and not isinstance(value, bool)
|
|
150
|
+
case "boolean":
|
|
151
|
+
return isinstance(value, bool)
|
|
152
|
+
case "null":
|
|
153
|
+
return value is None
|
|
154
|
+
return True
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def minimal_validate(schema: Any, value: Any, path: str = "$") -> list[str]:
|
|
158
|
+
"""A small JSON-schema subset validator used when ``jsonschema`` is not installed.
|
|
159
|
+
|
|
160
|
+
Supports: type (string or list), enum, const, properties, required,
|
|
161
|
+
additionalProperties (bool), items, minItems/maxItems, minLength/maxLength,
|
|
162
|
+
minimum/maximum, pattern, anyOf/oneOf/allOf, not. Returns error strings.
|
|
163
|
+
"""
|
|
164
|
+
if schema is True or schema == {}:
|
|
165
|
+
return []
|
|
166
|
+
if schema is False:
|
|
167
|
+
return [f"{path}: schema forbids any value"]
|
|
168
|
+
if not isinstance(schema, dict):
|
|
169
|
+
return [f"{path}: unsupported schema {schema!r}"]
|
|
170
|
+
errors: list[str] = []
|
|
171
|
+
|
|
172
|
+
if "type" in schema:
|
|
173
|
+
types = schema["type"] if isinstance(schema["type"], list) else [schema["type"]]
|
|
174
|
+
if not any(_type_matches(value, t) for t in types):
|
|
175
|
+
errors.append(f"{path}: expected type {'/'.join(types)}, got {type(value).__name__}")
|
|
176
|
+
return errors
|
|
177
|
+
if "enum" in schema and value not in schema["enum"]:
|
|
178
|
+
errors.append(f"{path}: {value!r} not in enum {schema['enum']!r}")
|
|
179
|
+
if "const" in schema and value != schema["const"]:
|
|
180
|
+
errors.append(f"{path}: expected const {schema['const']!r}")
|
|
181
|
+
|
|
182
|
+
if isinstance(value, dict):
|
|
183
|
+
for key in schema.get("required", []) or []:
|
|
184
|
+
if key not in value:
|
|
185
|
+
errors.append(f"{path}: missing required property {key!r}")
|
|
186
|
+
props = schema.get("properties", {}) or {}
|
|
187
|
+
for key, sub in props.items():
|
|
188
|
+
if key in value:
|
|
189
|
+
errors.extend(minimal_validate(sub, value[key], f"{path}.{key}"))
|
|
190
|
+
if schema.get("additionalProperties") is False:
|
|
191
|
+
extra = sorted(set(value) - set(props))
|
|
192
|
+
if extra:
|
|
193
|
+
errors.append(f"{path}: additional properties not allowed: {', '.join(extra)}")
|
|
194
|
+
if isinstance(value, list):
|
|
195
|
+
if "minItems" in schema and len(value) < schema["minItems"]:
|
|
196
|
+
errors.append(f"{path}: fewer than {schema['minItems']} items")
|
|
197
|
+
if "maxItems" in schema and len(value) > schema["maxItems"]:
|
|
198
|
+
errors.append(f"{path}: more than {schema['maxItems']} items")
|
|
199
|
+
items = schema.get("items")
|
|
200
|
+
if isinstance(items, dict | bool):
|
|
201
|
+
for i, item in enumerate(value):
|
|
202
|
+
errors.extend(minimal_validate(items, item, f"{path}[{i}]"))
|
|
203
|
+
if isinstance(value, str):
|
|
204
|
+
if "minLength" in schema and len(value) < schema["minLength"]:
|
|
205
|
+
errors.append(f"{path}: shorter than {schema['minLength']}")
|
|
206
|
+
if "maxLength" in schema and len(value) > schema["maxLength"]:
|
|
207
|
+
errors.append(f"{path}: longer than {schema['maxLength']}")
|
|
208
|
+
if "pattern" in schema and re.search(schema["pattern"], value) is None:
|
|
209
|
+
errors.append(f"{path}: does not match pattern {schema['pattern']!r}")
|
|
210
|
+
if isinstance(value, int | float) and not isinstance(value, bool):
|
|
211
|
+
if "minimum" in schema and value < schema["minimum"]:
|
|
212
|
+
errors.append(f"{path}: {value} < minimum {schema['minimum']}")
|
|
213
|
+
if "maximum" in schema and value > schema["maximum"]:
|
|
214
|
+
errors.append(f"{path}: {value} > maximum {schema['maximum']}")
|
|
215
|
+
|
|
216
|
+
if "allOf" in schema:
|
|
217
|
+
for sub in schema["allOf"]:
|
|
218
|
+
errors.extend(minimal_validate(sub, value, path))
|
|
219
|
+
if "anyOf" in schema and not any(not minimal_validate(s, value, path) for s in schema["anyOf"]):
|
|
220
|
+
errors.append(f"{path}: matches none of anyOf")
|
|
221
|
+
if "oneOf" in schema:
|
|
222
|
+
matches = sum(1 for s in schema["oneOf"] if not minimal_validate(s, value, path))
|
|
223
|
+
if matches != 1:
|
|
224
|
+
errors.append(f"{path}: matches {matches} of oneOf, expected exactly 1")
|
|
225
|
+
if "not" in schema and not minimal_validate(schema["not"], value, path):
|
|
226
|
+
errors.append(f"{path}: matches forbidden schema")
|
|
227
|
+
return errors
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
def validate_json_schema(schema: Any, value: Any) -> list[str]:
|
|
231
|
+
"""Validate with ``jsonschema`` when installed, else the minimal validator."""
|
|
232
|
+
try:
|
|
233
|
+
import jsonschema
|
|
234
|
+
except ImportError:
|
|
235
|
+
return minimal_validate(schema, value)
|
|
236
|
+
validator_cls = jsonschema.validators.validator_for(schema)
|
|
237
|
+
validator = validator_cls(schema)
|
|
238
|
+
return [
|
|
239
|
+
f"$.{'.'.join(str(p) for p in err.absolute_path)}: {err.message}"
|
|
240
|
+
if err.absolute_path
|
|
241
|
+
else f"$: {err.message}"
|
|
242
|
+
for err in sorted(validator.iter_errors(value), key=lambda e: list(e.absolute_path))
|
|
243
|
+
]
|
|
244
|
+
|
|
245
|
+
|
|
246
|
+
# How many `{` / `[` positions the search for a reply's JSON tries (a guard against
|
|
247
|
+
# pathological replies, such as thousands of unmatched brackets).
|
|
248
|
+
MAX_JSON_STARTS = 2000
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
def _json_values(text: str) -> list[Any]:
|
|
252
|
+
"""Every top-level JSON object or array in ``text``, in order (inside code fences too).
|
|
253
|
+
|
|
254
|
+
Each ``{`` or ``[`` not inside a value found already is tried as the start
|
|
255
|
+
of one; prose around and between them is skipped.
|
|
256
|
+
"""
|
|
257
|
+
decoder = json.JSONDecoder()
|
|
258
|
+
values: list[Any] = []
|
|
259
|
+
position = 0
|
|
260
|
+
for _ in range(MAX_JSON_STARTS):
|
|
261
|
+
starts = [i for i in (text.find("{", position), text.find("[", position)) if i >= 0]
|
|
262
|
+
if not starts:
|
|
263
|
+
break
|
|
264
|
+
start = min(starts)
|
|
265
|
+
try:
|
|
266
|
+
value, end = decoder.raw_decode(text, start)
|
|
267
|
+
except json.JSONDecodeError:
|
|
268
|
+
position = start + 1
|
|
269
|
+
continue
|
|
270
|
+
values.append(value)
|
|
271
|
+
position = end
|
|
272
|
+
return values
|
|
273
|
+
|
|
274
|
+
|
|
275
|
+
def _root_types(schema: Any) -> tuple[type, ...] | None:
|
|
276
|
+
"""The Python types of the schema's root ``type`` (object or array), or None for any."""
|
|
277
|
+
declared = schema.get("type") if isinstance(schema, dict) else None
|
|
278
|
+
names = declared if isinstance(declared, list) else [declared]
|
|
279
|
+
types = tuple(t for n in names for t in {"object": (dict,), "array": (list,)}.get(n, ()))
|
|
280
|
+
return types or None
|
|
281
|
+
|
|
282
|
+
|
|
283
|
+
def _extract_json(response: str, schema: Any = None) -> Any:
|
|
284
|
+
"""The JSON answer in a reply: the whole reply when it is JSON, else the reply's last JSON
|
|
285
|
+
object or array (of the schema's root type, when it names object or array).
|
|
286
|
+
|
|
287
|
+
The last one, not the first: a reply often shows an example, or quotes its
|
|
288
|
+
input, before the answer, and may add prose after it.
|
|
289
|
+
"""
|
|
290
|
+
text = response.strip()
|
|
291
|
+
try:
|
|
292
|
+
return json.loads(text)
|
|
293
|
+
except json.JSONDecodeError:
|
|
294
|
+
pass
|
|
295
|
+
values = _json_values(text)
|
|
296
|
+
wanted = _root_types(schema)
|
|
297
|
+
if wanted is not None:
|
|
298
|
+
values = [v for v in values if isinstance(v, wanted)] or values
|
|
299
|
+
if values:
|
|
300
|
+
return values[-1]
|
|
301
|
+
raise json.JSONDecodeError("no JSON object or array found", text, 0)
|
|
302
|
+
|
|
303
|
+
|
|
304
|
+
def check_json_schema(schema: Any, response: str, structured: Any = None) -> CheckResult:
|
|
305
|
+
"""``structured``: the run's structured response (a project with a response schema),
|
|
306
|
+
checked as it is; without one, the JSON in ``response`` (``_extract_json``)."""
|
|
307
|
+
if structured is not None:
|
|
308
|
+
errors = validate_json_schema(schema, structured)
|
|
309
|
+
if errors:
|
|
310
|
+
return False, "structured response: schema violation: " + "; ".join(errors[:3])
|
|
311
|
+
return True, ""
|
|
312
|
+
try:
|
|
313
|
+
parsed = _extract_json(response, schema)
|
|
314
|
+
except json.JSONDecodeError as exc:
|
|
315
|
+
return False, f"response is not valid JSON: {exc.msg}"
|
|
316
|
+
errors = validate_json_schema(schema, parsed)
|
|
317
|
+
if errors:
|
|
318
|
+
return False, "schema violation: " + "; ".join(errors[:3])
|
|
319
|
+
return True, ""
|
|
320
|
+
|
|
321
|
+
|
|
322
|
+
# --- tool calls -------------------------------------------------------------
|
|
323
|
+
|
|
324
|
+
|
|
325
|
+
def args_subset_matches(subset: dict[str, Any] | None, args: Any) -> bool:
|
|
326
|
+
"""True when every key in ``subset`` equals the same key in ``args`` (recursively)."""
|
|
327
|
+
if not subset:
|
|
328
|
+
return True
|
|
329
|
+
if not isinstance(args, dict):
|
|
330
|
+
return False
|
|
331
|
+
for key, expected in subset.items():
|
|
332
|
+
if key not in args:
|
|
333
|
+
return False
|
|
334
|
+
actual = args[key]
|
|
335
|
+
if isinstance(expected, dict) and isinstance(actual, dict):
|
|
336
|
+
if not args_subset_matches(expected, actual):
|
|
337
|
+
return False
|
|
338
|
+
elif expected != actual:
|
|
339
|
+
return False
|
|
340
|
+
return True
|
|
341
|
+
|
|
342
|
+
|
|
343
|
+
def _describe_call(call: dict[str, Any]) -> str:
|
|
344
|
+
subset = call.get("args_subset")
|
|
345
|
+
return f"{call.get('name')}({json.dumps(subset, sort_keys=True) if subset else ''})"
|
|
346
|
+
|
|
347
|
+
|
|
348
|
+
def check_tool_calls(
|
|
349
|
+
expected: list[dict[str, Any]], actual: list[dict[str, Any]], *, ordered: bool = False
|
|
350
|
+
) -> CheckResult:
|
|
351
|
+
"""Every expected call matches a distinct actual call; ordered = as a subsequence."""
|
|
352
|
+
actual_names = [str(c.get("name")) for c in actual]
|
|
353
|
+
if ordered:
|
|
354
|
+
position = 0
|
|
355
|
+
for exp in expected:
|
|
356
|
+
while position < len(actual):
|
|
357
|
+
cand = actual[position]
|
|
358
|
+
position += 1
|
|
359
|
+
if cand.get("name") == exp.get("name") and args_subset_matches(
|
|
360
|
+
exp.get("args_subset"), cand.get("args")
|
|
361
|
+
):
|
|
362
|
+
break
|
|
363
|
+
else:
|
|
364
|
+
return False, (
|
|
365
|
+
f"expected tool call {_describe_call(exp)} not found in order; "
|
|
366
|
+
f"actual calls: {actual_names or 'none'}"
|
|
367
|
+
)
|
|
368
|
+
return True, ""
|
|
369
|
+
|
|
370
|
+
remaining = list(actual)
|
|
371
|
+
for exp in expected:
|
|
372
|
+
for i, cand in enumerate(remaining):
|
|
373
|
+
if cand.get("name") == exp.get("name") and args_subset_matches(
|
|
374
|
+
exp.get("args_subset"), cand.get("args")
|
|
375
|
+
):
|
|
376
|
+
del remaining[i]
|
|
377
|
+
break
|
|
378
|
+
else:
|
|
379
|
+
return False, (
|
|
380
|
+
f"expected tool call {_describe_call(exp)} not found; "
|
|
381
|
+
f"actual calls: {actual_names or 'none'}"
|
|
382
|
+
)
|
|
383
|
+
return True, ""
|
|
384
|
+
|
|
385
|
+
|
|
386
|
+
# An `expect.approvals` status -> the recorded gate statuses it accepts.
|
|
387
|
+
_APPROVAL_OUTCOMES: dict[str, tuple[str, ...] | None] = {
|
|
388
|
+
"gated": None, # the call hit the gate, however it ended
|
|
389
|
+
"approved": ("approved",),
|
|
390
|
+
"rejected": ("rejected",),
|
|
391
|
+
}
|
|
392
|
+
|
|
393
|
+
|
|
394
|
+
def _describe_approval(record: dict[str, Any]) -> str:
|
|
395
|
+
call = f"{record.get('method') or '?'} {record.get('path') or '?'}"
|
|
396
|
+
if record.get("operation_id"):
|
|
397
|
+
call = f"{record['operation_id']} {call}"
|
|
398
|
+
return f"{call} ({record.get('status') or 'undecided'})"
|
|
399
|
+
|
|
400
|
+
|
|
401
|
+
def check_approvals(expected: list[dict[str, Any]], actual: list[dict[str, Any]]) -> CheckResult:
|
|
402
|
+
"""Every expected gate matches a distinct recorded one with an accepted status."""
|
|
403
|
+
remaining = list(actual)
|
|
404
|
+
for exp in expected:
|
|
405
|
+
status = exp.get("status", "gated")
|
|
406
|
+
accepted = _APPROVAL_OUTCOMES.get(status)
|
|
407
|
+
for i, record in enumerate(remaining):
|
|
408
|
+
if call_matches(exp["match"], record) and (
|
|
409
|
+
accepted is None or record.get("status") in accepted
|
|
410
|
+
):
|
|
411
|
+
del remaining[i]
|
|
412
|
+
break
|
|
413
|
+
else:
|
|
414
|
+
recorded = ", ".join(_describe_approval(r) for r in actual) or "none"
|
|
415
|
+
return False, (
|
|
416
|
+
f"expected {describe_match(exp['match'])} to be {status} at an approval gate; "
|
|
417
|
+
f"gates hit: {recorded}"
|
|
418
|
+
)
|
|
419
|
+
return True, ""
|
|
420
|
+
|
|
421
|
+
|
|
422
|
+
def check_no_approvals(actual: list[dict[str, Any]]) -> CheckResult:
|
|
423
|
+
if actual:
|
|
424
|
+
return False, "calls hit an approval gate: " + ", ".join(
|
|
425
|
+
_describe_approval(r) for r in actual
|
|
426
|
+
)
|
|
427
|
+
return True, ""
|
|
428
|
+
|
|
429
|
+
|
|
430
|
+
def check_no_tool_calls(actual: list[dict[str, Any]]) -> CheckResult:
|
|
431
|
+
if actual:
|
|
432
|
+
return False, "unexpected tool calls: " + ", ".join(str(c.get("name")) for c in actual)
|
|
433
|
+
return True, ""
|
|
434
|
+
|
|
435
|
+
|
|
436
|
+
def check_max_latency_ms(
|
|
437
|
+
limit: float, latency_ms: float | list[float | None] | None
|
|
438
|
+
) -> CheckResult:
|
|
439
|
+
"""Latency at or below ``limit``; given every turn's latency, each turn must be."""
|
|
440
|
+
if isinstance(latency_ms, list):
|
|
441
|
+
for index, value in enumerate(latency_ms, start=1):
|
|
442
|
+
passed, reason = check_max_latency_ms(limit, value)
|
|
443
|
+
if not passed:
|
|
444
|
+
return False, f"turn {index}: {reason}"
|
|
445
|
+
return True, ""
|
|
446
|
+
if latency_ms is None:
|
|
447
|
+
return False, "trace has no latency_ms"
|
|
448
|
+
if latency_ms > limit:
|
|
449
|
+
return False, f"latency {latency_ms:.0f} ms exceeds {limit:.0f} ms"
|
|
450
|
+
return True, ""
|
|
451
|
+
|
|
452
|
+
|
|
453
|
+
def total_tokens(usage: dict[str, Any] | None) -> int | None:
|
|
454
|
+
if not isinstance(usage, dict):
|
|
455
|
+
return None
|
|
456
|
+
if isinstance(usage.get("total_tokens"), int | float):
|
|
457
|
+
return int(usage["total_tokens"])
|
|
458
|
+
parts = [usage.get("input_tokens"), usage.get("output_tokens")]
|
|
459
|
+
if all(isinstance(p, int | float) for p in parts):
|
|
460
|
+
return int(parts[0]) + int(parts[1])
|
|
461
|
+
return None
|
|
462
|
+
|
|
463
|
+
|
|
464
|
+
def check_max_tokens(
|
|
465
|
+
limit: float, usage: dict[str, Any] | list[dict[str, Any] | None] | None
|
|
466
|
+
) -> CheckResult:
|
|
467
|
+
"""Tokens at or below ``limit``; given every turn's usage, their sum must be."""
|
|
468
|
+
if isinstance(usage, list):
|
|
469
|
+
totals = [total_tokens(u) for u in usage]
|
|
470
|
+
if not totals or any(t is None for t in totals):
|
|
471
|
+
return False, "a turn has no usage.input_tokens/output_tokens"
|
|
472
|
+
total = sum(t for t in totals if t is not None)
|
|
473
|
+
if total > limit:
|
|
474
|
+
return False, f"{total} tokens over {len(totals)} turns exceed {limit:.0f}"
|
|
475
|
+
return True, ""
|
|
476
|
+
total = total_tokens(usage)
|
|
477
|
+
if total is None:
|
|
478
|
+
return False, "trace has no usage.input_tokens/output_tokens"
|
|
479
|
+
if total > limit:
|
|
480
|
+
return False, f"{total} tokens exceed {limit:.0f}"
|
|
481
|
+
return True, ""
|
|
482
|
+
|
|
483
|
+
|
|
484
|
+
# --- driver -----------------------------------------------------------------
|
|
485
|
+
|
|
486
|
+
|
|
487
|
+
def declared_checks(expect: dict[str, Any]) -> list[str]:
|
|
488
|
+
"""The check names a case's ``expect`` block activates (``ordered`` is a modifier)."""
|
|
489
|
+
names: list[str] = []
|
|
490
|
+
if expect.get("contains"):
|
|
491
|
+
names.append("contains")
|
|
492
|
+
if expect.get("not_contains"):
|
|
493
|
+
names.append("not_contains")
|
|
494
|
+
if expect.get("regex") is not None:
|
|
495
|
+
names.append("regex")
|
|
496
|
+
if expect.get("json_schema") is not None:
|
|
497
|
+
names.append("json_schema")
|
|
498
|
+
if expect.get("tool_calls") is not None:
|
|
499
|
+
names.append("tool_calls")
|
|
500
|
+
if expect.get("no_tool_calls"):
|
|
501
|
+
names.append("no_tool_calls")
|
|
502
|
+
if expect.get("max_latency_ms") is not None:
|
|
503
|
+
names.append("max_latency_ms")
|
|
504
|
+
if expect.get("max_tokens") is not None:
|
|
505
|
+
names.append("max_tokens")
|
|
506
|
+
if expect.get("approvals") is not None:
|
|
507
|
+
names.append("approvals")
|
|
508
|
+
if expect.get("no_approvals"):
|
|
509
|
+
names.append("no_approvals")
|
|
510
|
+
return names
|
|
511
|
+
|
|
512
|
+
|
|
513
|
+
def _all_turns(trace: dict[str, Any]) -> list[dict[str, Any]] | None:
|
|
514
|
+
"""Every turn record of a multi-turn trace, or None for a single-turn one."""
|
|
515
|
+
turns = trace.get("turns")
|
|
516
|
+
if not isinstance(turns, list) or not turns:
|
|
517
|
+
return None
|
|
518
|
+
return [t if isinstance(t, dict) else {} for t in turns]
|
|
519
|
+
|
|
520
|
+
|
|
521
|
+
def _text(value: Any) -> str:
|
|
522
|
+
return "" if value is None else str(value)
|
|
523
|
+
|
|
524
|
+
|
|
525
|
+
def run_checks(expect: dict[str, Any], trace: dict[str, Any]) -> dict[str, dict[str, Any]]:
|
|
526
|
+
"""Apply the declared checks to ``trace``; returns ``{name: {passed, reason}}``.
|
|
527
|
+
|
|
528
|
+
``expect.scope`` picks what they read on a multi-turn case: the final turn
|
|
529
|
+
(``final_turn``, the default; the trace's top-level fields) or every turn
|
|
530
|
+
(``all_turns``: every reply, every tool call in order, each turn's latency,
|
|
531
|
+
the summed tokens, every approval gate). ``json_schema`` always reads the
|
|
532
|
+
final answer: the structured response when the trace has one, else the
|
|
533
|
+
final reply.
|
|
534
|
+
|
|
535
|
+
A check that raises is reported as failed with the exception text, so a bad
|
|
536
|
+
expectation never aborts grading.
|
|
537
|
+
"""
|
|
538
|
+
final_response = _text(trace.get("response"))
|
|
539
|
+
turns = _all_turns(trace) if expect.get("scope") == SCOPE_ALL_TURNS else None
|
|
540
|
+
response: str | list[str]
|
|
541
|
+
latency: Any
|
|
542
|
+
usage: Any
|
|
543
|
+
if turns is None:
|
|
544
|
+
response = final_response
|
|
545
|
+
tool_calls = trace.get("tool_calls") or []
|
|
546
|
+
latency = trace.get("latency_ms")
|
|
547
|
+
usage = trace.get("usage")
|
|
548
|
+
approvals = trace.get("approvals") or []
|
|
549
|
+
else:
|
|
550
|
+
response = [_text(t.get("response")) for t in turns]
|
|
551
|
+
tool_calls = [c for t in turns for c in (t.get("tool_calls") or [])]
|
|
552
|
+
latency = [t.get("latency_ms") for t in turns]
|
|
553
|
+
usage = [t.get("usage") for t in turns]
|
|
554
|
+
approvals = [a for t in turns for a in (t.get("approvals") or [])]
|
|
555
|
+
fold = bool(expect.get("case_insensitive", True))
|
|
556
|
+
runners: dict[str, Callable[[], CheckResult]] = {
|
|
557
|
+
"contains": lambda: check_contains(expect["contains"], response, case_insensitive=fold),
|
|
558
|
+
"not_contains": lambda: check_not_contains(
|
|
559
|
+
expect["not_contains"], response, case_insensitive=fold
|
|
560
|
+
),
|
|
561
|
+
"regex": lambda: check_regex(expect["regex"], response),
|
|
562
|
+
"json_schema": lambda: check_json_schema(
|
|
563
|
+
expect["json_schema"], final_response, trace.get("structured_response")
|
|
564
|
+
),
|
|
565
|
+
"tool_calls": lambda: check_tool_calls(
|
|
566
|
+
expect["tool_calls"], tool_calls, ordered=bool(expect.get("ordered"))
|
|
567
|
+
),
|
|
568
|
+
"no_tool_calls": lambda: check_no_tool_calls(tool_calls),
|
|
569
|
+
"max_latency_ms": lambda: check_max_latency_ms(expect["max_latency_ms"], latency),
|
|
570
|
+
"max_tokens": lambda: check_max_tokens(expect["max_tokens"], usage),
|
|
571
|
+
"approvals": lambda: check_approvals(expect["approvals"], approvals),
|
|
572
|
+
"no_approvals": lambda: check_no_approvals(approvals),
|
|
573
|
+
}
|
|
574
|
+
results: dict[str, dict[str, Any]] = {}
|
|
575
|
+
for name in declared_checks(expect):
|
|
576
|
+
try:
|
|
577
|
+
passed, reason = runners[name]()
|
|
578
|
+
except Exception as exc: # a malformed expectation must not abort grading
|
|
579
|
+
passed, reason = False, f"check raised {type(exc).__name__}: {_short(str(exc))}"
|
|
580
|
+
results[name] = {"passed": bool(passed), "reason": reason}
|
|
581
|
+
return results
|