graph-agents-cli 0.3.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- graph_agents_cli/__init__.py +26 -0
- graph_agents_cli/_api_policy.py +2145 -0
- graph_agents_cli/_approvals.py +400 -0
- graph_agents_cli/_build.py +186 -0
- graph_agents_cli/_build_info.json +7 -0
- graph_agents_cli/_chat_client.py +462 -0
- graph_agents_cli/_click.py +157 -0
- graph_agents_cli/_defaults.py +139 -0
- graph_agents_cli/_experiments.py +64 -0
- graph_agents_cli/_http.py +192 -0
- graph_agents_cli/_output.py +83 -0
- graph_agents_cli/_project.py +462 -0
- graph_agents_cli/_remote.py +220 -0
- graph_agents_cli/_response_schema.py +264 -0
- graph_agents_cli/_runner.py +319 -0
- graph_agents_cli/_skills_check.py +274 -0
- graph_agents_cli/_tools.py +189 -0
- graph_agents_cli/_trust.py +66 -0
- graph_agents_cli/api/__init__.py +15 -0
- graph_agents_cli/api/_changes.py +506 -0
- graph_agents_cli/api/_files.py +658 -0
- graph_agents_cli/api/cmd_api.py +2480 -0
- graph_agents_cli/deploy/__init__.py +15 -0
- graph_agents_cli/deploy/_config.py +171 -0
- graph_agents_cli/deploy/_image.py +128 -0
- graph_agents_cli/deploy/_kube.py +286 -0
- graph_agents_cli/deploy/_modes.py +234 -0
- graph_agents_cli/deploy/_preflight.py +370 -0
- graph_agents_cli/deploy/_values.py +168 -0
- graph_agents_cli/deploy/cmd_deploy.py +1866 -0
- graph_agents_cli/deploy/gitops.py +562 -0
- graph_agents_cli/deploy/local_load.py +273 -0
- graph_agents_cli/dev/__init__.py +13 -0
- graph_agents_cli/dev/cmd_build.py +131 -0
- graph_agents_cli/dev/cmd_install.py +78 -0
- graph_agents_cli/dev/cmd_lint.py +119 -0
- graph_agents_cli/dev/cmd_playground.py +297 -0
- graph_agents_cli/dev/policy_check.py +1287 -0
- graph_agents_cli/eval/__init__.py +22 -0
- graph_agents_cli/eval/_client.py +670 -0
- graph_agents_cli/eval/_common.py +177 -0
- graph_agents_cli/eval/_judge.py +168 -0
- graph_agents_cli/eval/_judge_runner.py +238 -0
- graph_agents_cli/eval/_paths.py +212 -0
- graph_agents_cli/eval/checks.py +581 -0
- graph_agents_cli/eval/cmd_analyze.py +278 -0
- graph_agents_cli/eval/cmd_compare.py +284 -0
- graph_agents_cli/eval/cmd_eval_group.py +80 -0
- graph_agents_cli/eval/cmd_generate.py +558 -0
- graph_agents_cli/eval/cmd_grade.py +466 -0
- graph_agents_cli/eval/cmd_metric.py +156 -0
- graph_agents_cli/eval/cmd_run.py +370 -0
- graph_agents_cli/eval/cmd_submit.py +400 -0
- graph_agents_cli/eval/config.py +435 -0
- graph_agents_cli/eval/dataset.py +350 -0
- graph_agents_cli/eval/gate.py +420 -0
- graph_agents_cli/eval/transcript.py +192 -0
- graph_agents_cli/extension/__init__.py +13 -0
- graph_agents_cli/extension/_compat.py +86 -0
- graph_agents_cli/extension/_loader.py +293 -0
- graph_agents_cli/extension/_manifest.py +135 -0
- graph_agents_cli/extension/_overrides.py +195 -0
- graph_agents_cli/extension/_paths.py +91 -0
- graph_agents_cli/extension/_refs.py +193 -0
- graph_agents_cli/extension/_resolver.py +453 -0
- graph_agents_cli/extension/_schema.py +106 -0
- graph_agents_cli/extension/_spec.py +253 -0
- graph_agents_cli/extension/_sync.py +102 -0
- graph_agents_cli/extension/_trust.py +58 -0
- graph_agents_cli/extension/cmd_extension_add.py +259 -0
- graph_agents_cli/extension/cmd_extension_group.py +57 -0
- graph_agents_cli/extension/cmd_extension_list.py +56 -0
- graph_agents_cli/extension/cmd_extension_remove.py +61 -0
- graph_agents_cli/extension/cmd_extension_update.py +195 -0
- graph_agents_cli/info/__init__.py +13 -0
- graph_agents_cli/info/cmd_info.py +222 -0
- graph_agents_cli/infra/__init__.py +15 -0
- graph_agents_cli/infra/checks.py +1169 -0
- graph_agents_cli/infra/cmd_infra.py +103 -0
- graph_agents_cli/main.py +591 -0
- graph_agents_cli/peer/__init__.py +15 -0
- graph_agents_cli/peer/_generate.py +254 -0
- graph_agents_cli/peer/cmd_peer.py +1151 -0
- graph_agents_cli/run/__init__.py +13 -0
- graph_agents_cli/run/_local_server.py +1157 -0
- graph_agents_cli/run/_signals.py +141 -0
- graph_agents_cli/run/cmd_approvals.py +530 -0
- graph_agents_cli/run/cmd_run.py +1421 -0
- graph_agents_cli/scaffold/__init__.py +19 -0
- graph_agents_cli/scaffold/agents/README.md +24 -0
- graph_agents_cli/scaffold/agents/empty_py/.template/templateconfig.yaml +22 -0
- graph_agents_cli/scaffold/agents/langgraph/.env.example +292 -0
- graph_agents_cli/scaffold/agents/langgraph/.template/templateconfig.yaml +28 -0
- graph_agents_cli/scaffold/agents/langgraph/Dockerfile +59 -0
- graph_agents_cli/scaffold/agents/langgraph/Dockerfile.langgraph-server +59 -0
- graph_agents_cli/scaffold/agents/langgraph/README.md +571 -0
- graph_agents_cli/scaffold/agents/langgraph/api-policy.yaml +60 -0
- graph_agents_cli/scaffold/agents/langgraph/app/__init__.py +20 -0
- graph_agents_cli/scaffold/agents/langgraph/app/agent.py +174 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/__init__.py +15 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/a2a.py +2162 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/a2a_client.py +1167 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/api_client.py +4220 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/approvals.py +1349 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/auth.py +1986 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/chat.py +2962 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/checkpointer.py +432 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/content.py +569 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/db.py +580 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/limits.py +203 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/metrics.py +231 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/middleware.py +361 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/model.py +611 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/playground.py +230 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/run_locks.py +459 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/structured.py +755 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/telemetry.py +681 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/threads.py +493 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/token_exchange.py +959 -0
- graph_agents_cli/scaffold/agents/langgraph/app/fast_api_app.py +770 -0
- graph_agents_cli/scaffold/agents/langgraph/app/policies/__init__.py +55 -0
- graph_agents_cli/scaffold/agents/langgraph/app/policies/custom.py +97 -0
- graph_agents_cli/scaffold/agents/langgraph/app/tools/__init__.py +46 -0
- graph_agents_cli/scaffold/agents/langgraph/app/tools/example_api.py +92 -0
- graph_agents_cli/scaffold/agents/langgraph/app/tools/weather.py +33 -0
- graph_agents_cli/scaffold/agents/langgraph/langgraph.json +14 -0
- graph_agents_cli/scaffold/agents/langgraph/pyproject.toml +78 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/conftest.py +376 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/eval/datasets/basic-dataset.json +53 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/eval/eval_config.yaml +32 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/approval_graph.py +137 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/fake_issuer.py +216 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/fake_openai.py +357 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_a2a_outcomes.py +569 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_a2a_relay.py +479 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_api_surface.py +812 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_approvals.py +1367 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_approvals_server.py +794 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_cross_actor.py +497 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_cross_actor_server.py +247 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_history_repair.py +278 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_model_apis.py +242 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_postgres.py +637 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_resilience_postgres.py +770 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_runtime_guardrails.py +854 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_server_e2e.py +340 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_server_runtime.py +989 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_structured_answers.py +584 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_structured_server.py +222 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_token_exchange_issuer.py +650 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/load_test/.results/.placeholder +0 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/load_test/README.md +22 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/load_test/conftest.py +21 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/load_test/load_test.py +81 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_a2a_client.py +824 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_a2a_scoping.py +724 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_api_client.py +1214 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_api_client_hardening.py +716 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_api_policy_rpc.py +767 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_approval_ledger.py +1536 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_fake_model.py +115 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_jwt_policy.py +991 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_limits.py +310 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_logging.py +148 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_logging_hardening.py +271 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_policy.py +378 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_resilience.py +610 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_server_auth.py +702 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_structured.py +673 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_telemetry.py +404 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_thread_listing.py +255 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_threads.py +268 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_token_exchange.py +1320 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_untrusted_content.py +393 -0
- graph_agents_cli/scaffold/agents/langgraph/uv-fastapi.lock +2084 -0
- graph_agents_cli/scaffold/agents/langgraph/uv-langgraph-server.lock +2106 -0
- graph_agents_cli/scaffold/agents/langgraph/{{cookiecutter.agent_guidance_filename}} +129 -0
- graph_agents_cli/scaffold/base_templates/_shared/graph-agents-cli-manifest.yaml +36 -0
- graph_agents_cli/scaffold/base_templates/python/.dockerignore +32 -0
- graph_agents_cli/scaffold/base_templates/python/.github/CODEOWNERS +30 -0
- graph_agents_cli/scaffold/base_templates/python/.github/agent.env +7 -0
- graph_agents_cli/scaffold/base_templates/python/.github/workflows/pr_checks.yaml +214 -0
- graph_agents_cli/scaffold/base_templates/python/.gitignore +209 -0
- graph_agents_cli/scaffold/base_templates/python/tests/unit/test_dummy.py +23 -0
- graph_agents_cli/scaffold/base_templates/python/{{cookiecutter.agent_guidance_filename}} +35 -0
- graph_agents_cli/scaffold/cmd_scaffold_group.py +49 -0
- graph_agents_cli/scaffold/commands/__init__.py +13 -0
- graph_agents_cli/scaffold/commands/create.py +1424 -0
- graph_agents_cli/scaffold/commands/enhance.py +1652 -0
- graph_agents_cli/scaffold/commands/upgrade.py +570 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/.github/agent.env +12 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/.github/workflows/promote-to-prod.yaml +371 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/.github/workflows/staging.yaml +450 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/argocd/application-dev.yaml +43 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/argocd/application-prod.yaml +41 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/argocd/application-staging.yaml +43 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/.helmignore +14 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/Chart.yaml +21 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/examples/networkpolicy.yaml +103 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/NOTES.txt +48 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/_helpers.tpl +189 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/certificate.yaml +15 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/configmap.yaml +10 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/deployment.yaml +199 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/hpa.yaml +22 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/httproute.yaml +30 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/ingress.yaml +39 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/networkpolicy.yaml +48 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/pdb.yaml +13 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/postgresql-secret.yaml +37 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/service.yaml +15 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/serviceaccount.yaml +13 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/servicemonitor.yaml +42 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values-dev.yaml +22 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values-prod.yaml +45 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values-staging.yaml +29 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values.yaml +396 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/tests/integration/test_chart.py +269 -0
- graph_agents_cli/scaffold/deployment_targets/none/README.md +5 -0
- graph_agents_cli/scaffold/deployment_targets/none/python/README.md +6 -0
- graph_agents_cli/scaffold/utils/__init__.py +13 -0
- graph_agents_cli/scaffold/utils/backup.py +212 -0
- graph_agents_cli/scaffold/utils/build_record.py +257 -0
- graph_agents_cli/scaffold/utils/cli_options.py +184 -0
- graph_agents_cli/scaffold/utils/fs.py +83 -0
- graph_agents_cli/scaffold/utils/generate_locks.py +214 -0
- graph_agents_cli/scaffold/utils/generation_metadata.py +88 -0
- graph_agents_cli/scaffold/utils/keyedit.py +768 -0
- graph_agents_cli/scaffold/utils/keymerge.py +537 -0
- graph_agents_cli/scaffold/utils/language.py +138 -0
- graph_agents_cli/scaffold/utils/lock_utils.py +94 -0
- graph_agents_cli/scaffold/utils/logging.py +77 -0
- graph_agents_cli/scaffold/utils/manifest.py +292 -0
- graph_agents_cli/scaffold/utils/merge.py +970 -0
- graph_agents_cli/scaffold/utils/merge3.py +216 -0
- graph_agents_cli/scaffold/utils/openapi_seed.py +199 -0
- graph_agents_cli/scaffold/utils/remote_template.py +376 -0
- graph_agents_cli/scaffold/utils/template.py +1352 -0
- graph_agents_cli/scaffold/utils/upgrade.py +894 -0
- graph_agents_cli/scaffold/utils/version.py +438 -0
- graph_agents_cli/secrets/__init__.py +15 -0
- graph_agents_cli/secrets/_apply.py +954 -0
- graph_agents_cli/secrets/_required.py +188 -0
- graph_agents_cli/secrets/cmd_secrets.py +211 -0
- graph_agents_cli/setup/__init__.py +13 -0
- graph_agents_cli/setup/_antigravity.py +221 -0
- graph_agents_cli/setup/cmd_auth.py +1030 -0
- graph_agents_cli/setup/cmd_dev_token.py +513 -0
- graph_agents_cli/setup/cmd_setup.py +428 -0
- graph_agents_cli/setup/cmd_update.py +140 -0
- graph_agents_cli/skills/__init__.py +13 -0
- graph_agents_cli/skills/_bundle.py +65 -0
- graph_agents_cli/skills/data/README.md +19 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/SKILL.md +357 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/references/github-settings.md +113 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/references/gitops.md +137 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/references/kubernetes.md +315 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/references/secrets.md +160 -0
- graph_agents_cli/skills/data/graph-agents-cli-eval/SKILL.md +303 -0
- graph_agents_cli/skills/data/graph-agents-cli-eval/references/dataset_schema.md +282 -0
- graph_agents_cli/skills/data/graph-agents-cli-eval/references/metrics-guide.md +143 -0
- graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/SKILL.md +659 -0
- graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/references/langchain-models.md +124 -0
- graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/references/langgraph.md +235 -0
- graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/references/template-contract.md +477 -0
- graph_agents_cli/skills/data/graph-agents-cli-observability/SKILL.md +231 -0
- graph_agents_cli/skills/data/graph-agents-cli-observability/references/langsmith.md +46 -0
- graph_agents_cli/skills/data/graph-agents-cli-observability/references/otel.md +59 -0
- graph_agents_cli/skills/data/graph-agents-cli-scaffold/SKILL.md +414 -0
- graph_agents_cli/skills/data/graph-agents-cli-scaffold/references/flags.md +134 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/SKILL.md +478 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/brainstorming.md +118 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/commands.md +419 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/extension.md +156 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/internals.md +67 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/spec-template.md +56 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/terminology.md +119 -0
- graph_agents_cli/system/__init__.py +15 -0
- graph_agents_cli/system/_apply.py +519 -0
- graph_agents_cli/system/_checks.py +1023 -0
- graph_agents_cli/system/_deploy.py +215 -0
- graph_agents_cli/system/_model.py +363 -0
- graph_agents_cli/system/_system.py +664 -0
- graph_agents_cli/system/_views.py +208 -0
- graph_agents_cli/system/cmd_system.py +423 -0
- graph_agents_cli-0.3.1.dist-info/METADATA +162 -0
- graph_agents_cli-0.3.1.dist-info/RECORD +291 -0
- graph_agents_cli-0.3.1.dist-info/WHEEL +4 -0
- graph_agents_cli-0.3.1.dist-info/entry_points.txt +2 -0
- graph_agents_cli-0.3.1.dist-info/licenses/LICENSE +201 -0
- graph_agents_cli-0.3.1.dist-info/licenses/NOTICE +19 -0
|
@@ -0,0 +1,435 @@
|
|
|
1
|
+
# Copyright 2026 graph-agents-cli contributors
|
|
2
|
+
#
|
|
3
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
4
|
+
# you may not use this file except in compliance with the License.
|
|
5
|
+
# You may obtain a copy of the License at
|
|
6
|
+
#
|
|
7
|
+
# https://www.apache.org/licenses/LICENSE-2.0
|
|
8
|
+
#
|
|
9
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
10
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
11
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
12
|
+
# See the License for the specific language governing permissions and
|
|
13
|
+
# limitations under the License.
|
|
14
|
+
|
|
15
|
+
"""``tests/eval/eval_config.yaml``: judge identity, quality metrics, rubrics.
|
|
16
|
+
|
|
17
|
+
Shape::
|
|
18
|
+
|
|
19
|
+
judge: { provider: null, model: null } # null = agent's provider/model
|
|
20
|
+
# max_tool_result_chars: 50000 (null = never cut)
|
|
21
|
+
quality_metrics: # only these may be below 100 percent
|
|
22
|
+
response_quality: { threshold: 4, min_pass_rate: 0.9 }
|
|
23
|
+
judges: # rubric text is versioned here
|
|
24
|
+
response_quality: { scale: 5, rubric: "...", prompt_template: "..." }
|
|
25
|
+
custom_metrics: [] # python callables: module:function
|
|
26
|
+
|
|
27
|
+
Built-in judges (``response_quality``, ``task_success``, ``groundedness``) work
|
|
28
|
+
without a ``judges:`` block; an entry there overrides any of their fields or
|
|
29
|
+
defines a new rubric.
|
|
30
|
+
"""
|
|
31
|
+
|
|
32
|
+
from __future__ import annotations
|
|
33
|
+
|
|
34
|
+
from dataclasses import dataclass, field
|
|
35
|
+
from pathlib import Path
|
|
36
|
+
from typing import Any
|
|
37
|
+
|
|
38
|
+
import yaml
|
|
39
|
+
|
|
40
|
+
from graph_agents_cli.eval._common import EvalConfigError
|
|
41
|
+
from graph_agents_cli.eval.transcript import DEFAULT_MAX_TOOL_RESULT_CHARS, render_tool_calls
|
|
42
|
+
|
|
43
|
+
DEFAULT_SCALE = 5
|
|
44
|
+
|
|
45
|
+
DEFAULT_PROMPT_TEMPLATE = """You are a strict, impartial evaluator of an AI agent's reply.
|
|
46
|
+
|
|
47
|
+
Rubric ({metric}):
|
|
48
|
+
{rubric}
|
|
49
|
+
|
|
50
|
+
Conversation so far (earlier turns in full: the user's message, each tool call the agent made \
|
|
51
|
+
with its result, and the agent's reply; then the user's latest message):
|
|
52
|
+
{conversation}
|
|
53
|
+
|
|
54
|
+
Agent response (the reply to the latest message; this is what you score):
|
|
55
|
+
{response}
|
|
56
|
+
{reference_section}{context_section}{tool_calls_section}
|
|
57
|
+
Score the agent response from 1 to {scale} against the rubric ({scale} is best), taking the \
|
|
58
|
+
earlier turns into account.
|
|
59
|
+
Reply with JSON only, on one line: {{"score": <number>, "reasoning": "<one or two sentences>"}}"""
|
|
60
|
+
|
|
61
|
+
# Placeholders a judge prompt_template may use; any other one is exit 3.
|
|
62
|
+
PROMPT_PLACEHOLDERS: tuple[str, ...] = (
|
|
63
|
+
"metric",
|
|
64
|
+
"rubric",
|
|
65
|
+
"scale",
|
|
66
|
+
"conversation",
|
|
67
|
+
"transcript",
|
|
68
|
+
"response",
|
|
69
|
+
"reference",
|
|
70
|
+
"context",
|
|
71
|
+
"reference_section",
|
|
72
|
+
"context_section",
|
|
73
|
+
"tool_calls_section",
|
|
74
|
+
)
|
|
75
|
+
|
|
76
|
+
JUDGE_KEYS: tuple[str, ...] = ("provider", "model", "max_tool_result_chars")
|
|
77
|
+
|
|
78
|
+
BUILTIN_JUDGES: dict[str, dict[str, Any]] = {
|
|
79
|
+
"response_quality": {
|
|
80
|
+
"description": "Accuracy, relevance and clarity of the final response.",
|
|
81
|
+
"scale": DEFAULT_SCALE,
|
|
82
|
+
"rubric": (
|
|
83
|
+
"Rate the response for accuracy, relevance to the user's request, and clarity. "
|
|
84
|
+
"5: fully correct, directly answers, clear and concise. 3: mostly correct but "
|
|
85
|
+
"incomplete, vague or verbose. 1: wrong, off-topic, or unusable. When a reference "
|
|
86
|
+
"answer is given, penalize factual disagreement with it."
|
|
87
|
+
),
|
|
88
|
+
},
|
|
89
|
+
"task_success": {
|
|
90
|
+
"description": "Whether the agent accomplished what the user asked for.",
|
|
91
|
+
"scale": DEFAULT_SCALE,
|
|
92
|
+
"rubric": (
|
|
93
|
+
"Judge whether the agent completed the user's task end to end, including using "
|
|
94
|
+
"tools when the task required it. In a multi-turn conversation the task is the "
|
|
95
|
+
"latest request in light of the earlier turns; what the agent already did or said "
|
|
96
|
+
"in an earlier turn counts and need not be repeated. 5: task fully accomplished "
|
|
97
|
+
"with a correct outcome. 3: partially accomplished or needs a follow-up from the "
|
|
98
|
+
"user. 1: task not attempted, abandoned, or completed incorrectly."
|
|
99
|
+
),
|
|
100
|
+
},
|
|
101
|
+
"groundedness": {
|
|
102
|
+
"description": "Whether every claim is supported by the supplied context or tool results.",
|
|
103
|
+
"scale": DEFAULT_SCALE,
|
|
104
|
+
"rubric": (
|
|
105
|
+
"Check every factual claim in the response against the supplied context and the "
|
|
106
|
+
"tool results of every turn, including earlier turns of the conversation. A tool "
|
|
107
|
+
"result marked TRUNCATED was cut for length: a claim that could come from its "
|
|
108
|
+
"omitted part is not unsupported. 5: every claim is supported. 3: mostly "
|
|
109
|
+
"supported with minor unsupported detail. 1: contradicts the context or invents "
|
|
110
|
+
"facts."
|
|
111
|
+
),
|
|
112
|
+
},
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
@dataclass
|
|
117
|
+
class JudgeSpec:
|
|
118
|
+
name: str
|
|
119
|
+
scale: float = DEFAULT_SCALE
|
|
120
|
+
rubric: str = ""
|
|
121
|
+
prompt_template: str = DEFAULT_PROMPT_TEMPLATE
|
|
122
|
+
description: str = ""
|
|
123
|
+
threshold: float | None = None
|
|
124
|
+
builtin: bool = False
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
@dataclass
|
|
128
|
+
class QualityMetric:
|
|
129
|
+
name: str
|
|
130
|
+
threshold: float
|
|
131
|
+
min_pass_rate: float = 1.0
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
@dataclass
|
|
135
|
+
class CustomMetric:
|
|
136
|
+
name: str
|
|
137
|
+
callable: str # "module:function", imported inside the project's environment
|
|
138
|
+
threshold: float = 1.0
|
|
139
|
+
description: str = ""
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
@dataclass
|
|
143
|
+
class EvalConfig:
|
|
144
|
+
judge_provider: str | None = None
|
|
145
|
+
judge_model: str | None = None
|
|
146
|
+
# Characters of one tool result shown to a judge; None = never cut.
|
|
147
|
+
max_tool_result_chars: int | None = DEFAULT_MAX_TOOL_RESULT_CHARS
|
|
148
|
+
quality_metrics: dict[str, QualityMetric] = field(default_factory=dict)
|
|
149
|
+
judges: dict[str, JudgeSpec] = field(default_factory=dict)
|
|
150
|
+
custom_metrics: dict[str, CustomMetric] = field(default_factory=dict)
|
|
151
|
+
source: Path | None = None
|
|
152
|
+
|
|
153
|
+
def metric_names(self) -> set[str]:
|
|
154
|
+
return set(self.judges) | set(self.custom_metrics)
|
|
155
|
+
|
|
156
|
+
def is_quality(self, metric: str) -> bool:
|
|
157
|
+
return metric in self.quality_metrics
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def builtin_judges() -> dict[str, JudgeSpec]:
|
|
161
|
+
return {
|
|
162
|
+
name: JudgeSpec(
|
|
163
|
+
name=name,
|
|
164
|
+
scale=spec["scale"],
|
|
165
|
+
rubric=spec["rubric"],
|
|
166
|
+
description=spec["description"],
|
|
167
|
+
builtin=True,
|
|
168
|
+
)
|
|
169
|
+
for name, spec in BUILTIN_JUDGES.items()
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
def _number(value: Any, where: str) -> float:
|
|
174
|
+
if isinstance(value, bool) or not isinstance(value, int | float):
|
|
175
|
+
raise EvalConfigError(f"{where} must be a number, got {value!r}")
|
|
176
|
+
return float(value)
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
def _parse_judges(raw: Any, where: str) -> dict[str, JudgeSpec]:
|
|
180
|
+
judges = builtin_judges()
|
|
181
|
+
if raw is None:
|
|
182
|
+
return judges
|
|
183
|
+
if not isinstance(raw, dict):
|
|
184
|
+
raise EvalConfigError(f"{where}: 'judges' must be a mapping of metric name to spec")
|
|
185
|
+
for name, spec in raw.items():
|
|
186
|
+
if spec is None:
|
|
187
|
+
spec = {}
|
|
188
|
+
if not isinstance(spec, dict):
|
|
189
|
+
raise EvalConfigError(f"{where}: judges.{name} must be a mapping")
|
|
190
|
+
base = judges.get(name) or JudgeSpec(name=str(name))
|
|
191
|
+
if "scale" in spec and spec["scale"] is not None:
|
|
192
|
+
base.scale = _number(spec["scale"], f"{where}: judges.{name}.scale")
|
|
193
|
+
if "rubric" in spec and spec["rubric"] is not None:
|
|
194
|
+
base.rubric = str(spec["rubric"])
|
|
195
|
+
if "prompt_template" in spec and spec["prompt_template"] is not None:
|
|
196
|
+
base.prompt_template = str(spec["prompt_template"])
|
|
197
|
+
if "description" in spec and spec["description"] is not None:
|
|
198
|
+
base.description = str(spec["description"])
|
|
199
|
+
if "threshold" in spec and spec["threshold"] is not None:
|
|
200
|
+
base.threshold = _number(spec["threshold"], f"{where}: judges.{name}.threshold")
|
|
201
|
+
if not base.rubric:
|
|
202
|
+
raise EvalConfigError(f"{where}: judges.{name} needs a 'rubric'")
|
|
203
|
+
_check_placeholders(base, where)
|
|
204
|
+
judges[str(name)] = base
|
|
205
|
+
return judges
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
def _check_placeholders(spec: JudgeSpec, where: str) -> None:
|
|
209
|
+
"""A template with an unknown placeholder fails when the config loads (exit 3).
|
|
210
|
+
|
|
211
|
+
``eval run`` loads the config before generating, so the mistake costs no
|
|
212
|
+
agent calls instead of surfacing at the first judged case.
|
|
213
|
+
"""
|
|
214
|
+
try:
|
|
215
|
+
spec.prompt_template.format(**dict.fromkeys(PROMPT_PLACEHOLDERS, ""))
|
|
216
|
+
except (KeyError, IndexError, ValueError) as exc:
|
|
217
|
+
raise EvalConfigError(
|
|
218
|
+
f"{where}: judges.{spec.name}.prompt_template has an unknown placeholder: {exc} "
|
|
219
|
+
f"(allowed: {', '.join('{' + p + '}' for p in PROMPT_PLACEHOLDERS)})"
|
|
220
|
+
) from exc
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
def _parse_custom_metrics(raw: Any, where: str) -> dict[str, CustomMetric]:
|
|
224
|
+
metrics: dict[str, CustomMetric] = {}
|
|
225
|
+
if raw is None:
|
|
226
|
+
return metrics
|
|
227
|
+
if not isinstance(raw, list):
|
|
228
|
+
raise EvalConfigError(f"{where}: 'custom_metrics' must be a list")
|
|
229
|
+
for i, item in enumerate(raw):
|
|
230
|
+
if isinstance(item, str):
|
|
231
|
+
item = {"callable": item}
|
|
232
|
+
if not isinstance(item, dict):
|
|
233
|
+
raise EvalConfigError(
|
|
234
|
+
f"{where}: custom_metrics[{i}] must be 'module:function' or a mapping"
|
|
235
|
+
)
|
|
236
|
+
target = item.get("callable") or item.get("function")
|
|
237
|
+
if not isinstance(target, str) or ":" not in target:
|
|
238
|
+
raise EvalConfigError(
|
|
239
|
+
f"{where}: custom_metrics[{i}] needs a 'callable' of the form module:function"
|
|
240
|
+
)
|
|
241
|
+
name = str(item.get("name") or target.rsplit(":", 1)[1])
|
|
242
|
+
threshold = item.get("threshold", 1.0)
|
|
243
|
+
metric = CustomMetric(
|
|
244
|
+
name=name,
|
|
245
|
+
callable=target,
|
|
246
|
+
threshold=_number(threshold, f"{where}: custom_metrics[{i}].threshold"),
|
|
247
|
+
description=str(item.get("description") or ""),
|
|
248
|
+
)
|
|
249
|
+
if name in metrics:
|
|
250
|
+
raise EvalConfigError(f"{where}: duplicate custom metric {name!r}")
|
|
251
|
+
metrics[name] = metric
|
|
252
|
+
return metrics
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
def _parse_quality_metrics(raw: Any, where: str, known: set[str]) -> dict[str, QualityMetric]:
|
|
256
|
+
metrics: dict[str, QualityMetric] = {}
|
|
257
|
+
if raw is None:
|
|
258
|
+
return metrics
|
|
259
|
+
if not isinstance(raw, dict):
|
|
260
|
+
raise EvalConfigError(f"{where}: 'quality_metrics' must be a mapping")
|
|
261
|
+
for name, spec in raw.items():
|
|
262
|
+
if spec is None:
|
|
263
|
+
spec = {}
|
|
264
|
+
if isinstance(spec, int | float) and not isinstance(spec, bool):
|
|
265
|
+
spec = {"threshold": spec}
|
|
266
|
+
if not isinstance(spec, dict):
|
|
267
|
+
raise EvalConfigError(f"{where}: quality_metrics.{name} must be a mapping")
|
|
268
|
+
if name not in known:
|
|
269
|
+
raise EvalConfigError(
|
|
270
|
+
f"{where}: quality metric {name!r} is not a known judge or custom metric "
|
|
271
|
+
f"(known: {', '.join(sorted(known))})"
|
|
272
|
+
)
|
|
273
|
+
if spec.get("threshold") is None:
|
|
274
|
+
raise EvalConfigError(f"{where}: quality metric {name!r} has no threshold")
|
|
275
|
+
threshold = _number(spec["threshold"], f"{where}: quality_metrics.{name}.threshold")
|
|
276
|
+
rate = spec.get("min_pass_rate", 1.0)
|
|
277
|
+
rate = _number(rate, f"{where}: quality_metrics.{name}.min_pass_rate")
|
|
278
|
+
if not 0.0 <= rate <= 1.0:
|
|
279
|
+
raise EvalConfigError(
|
|
280
|
+
f"{where}: quality_metrics.{name}.min_pass_rate must be between 0 and 1"
|
|
281
|
+
)
|
|
282
|
+
metrics[str(name)] = QualityMetric(name=str(name), threshold=threshold, min_pass_rate=rate)
|
|
283
|
+
return metrics
|
|
284
|
+
|
|
285
|
+
|
|
286
|
+
def _max_tool_result_chars(judge: dict[str, Any], where: str) -> int | None:
|
|
287
|
+
"""``judge.max_tool_result_chars``: absent = the default, null = never cut, else >= 1."""
|
|
288
|
+
if "max_tool_result_chars" not in judge:
|
|
289
|
+
return DEFAULT_MAX_TOOL_RESULT_CHARS
|
|
290
|
+
value = judge["max_tool_result_chars"]
|
|
291
|
+
if value is None:
|
|
292
|
+
return None
|
|
293
|
+
if isinstance(value, bool) or not isinstance(value, int) or value < 1:
|
|
294
|
+
raise EvalConfigError(
|
|
295
|
+
f"{where}: judge.max_tool_result_chars must be a whole number of characters "
|
|
296
|
+
f">= 1, or null to never cut tool results (got {value!r})"
|
|
297
|
+
)
|
|
298
|
+
return value
|
|
299
|
+
|
|
300
|
+
|
|
301
|
+
def parse_eval_config(data: Any, source: Path | None = None) -> EvalConfig:
|
|
302
|
+
"""Validate a parsed YAML mapping into an :class:`EvalConfig`."""
|
|
303
|
+
where = str(source) if source else "eval config"
|
|
304
|
+
if data is None:
|
|
305
|
+
data = {}
|
|
306
|
+
if not isinstance(data, dict):
|
|
307
|
+
raise EvalConfigError(f"{where}: top level must be a mapping")
|
|
308
|
+
unknown = sorted(set(data) - {"judge", "quality_metrics", "judges", "custom_metrics"})
|
|
309
|
+
if unknown:
|
|
310
|
+
raise EvalConfigError(f"{where}: unknown top-level key(s) {', '.join(unknown)}")
|
|
311
|
+
|
|
312
|
+
judge = data.get("judge") or {}
|
|
313
|
+
if not isinstance(judge, dict):
|
|
314
|
+
raise EvalConfigError(f"{where}: 'judge' must be a mapping with provider/model")
|
|
315
|
+
unknown_judge = sorted(str(k) for k in set(judge) - set(JUDGE_KEYS))
|
|
316
|
+
if unknown_judge:
|
|
317
|
+
raise EvalConfigError(
|
|
318
|
+
f"{where}: unknown judge key(s) {', '.join(unknown_judge)}; "
|
|
319
|
+
f"known: {', '.join(JUDGE_KEYS)}"
|
|
320
|
+
)
|
|
321
|
+
max_chars = _max_tool_result_chars(judge, where)
|
|
322
|
+
judges = _parse_judges(data.get("judges"), where)
|
|
323
|
+
custom = _parse_custom_metrics(data.get("custom_metrics"), where)
|
|
324
|
+
overlap = set(judges) & set(custom)
|
|
325
|
+
if overlap:
|
|
326
|
+
raise EvalConfigError(
|
|
327
|
+
f"{where}: {', '.join(sorted(overlap))} defined both as judge and custom metric"
|
|
328
|
+
)
|
|
329
|
+
quality = _parse_quality_metrics(data.get("quality_metrics"), where, set(judges) | set(custom))
|
|
330
|
+
provider = judge.get("provider")
|
|
331
|
+
model = judge.get("model")
|
|
332
|
+
return EvalConfig(
|
|
333
|
+
judge_provider=str(provider) if provider else None,
|
|
334
|
+
judge_model=str(model) if model else None,
|
|
335
|
+
max_tool_result_chars=max_chars,
|
|
336
|
+
quality_metrics=quality,
|
|
337
|
+
judges=judges,
|
|
338
|
+
custom_metrics=custom,
|
|
339
|
+
source=source,
|
|
340
|
+
)
|
|
341
|
+
|
|
342
|
+
|
|
343
|
+
def load_eval_config(path: Path | None, *, required: bool = False) -> EvalConfig:
|
|
344
|
+
"""Load ``eval_config.yaml``; a missing optional file yields the defaults."""
|
|
345
|
+
if path is None:
|
|
346
|
+
return parse_eval_config({}, None)
|
|
347
|
+
path = Path(path)
|
|
348
|
+
if not path.is_file():
|
|
349
|
+
if required:
|
|
350
|
+
raise EvalConfigError(f"eval config not found: {path}")
|
|
351
|
+
return parse_eval_config({}, None)
|
|
352
|
+
try:
|
|
353
|
+
data = yaml.safe_load(path.read_text(encoding="utf-8"))
|
|
354
|
+
except (OSError, yaml.YAMLError) as exc:
|
|
355
|
+
raise EvalConfigError(f"eval config {path} is not valid YAML: {exc}") from exc
|
|
356
|
+
return parse_eval_config(data, path)
|
|
357
|
+
|
|
358
|
+
|
|
359
|
+
def resolve_threshold(config: EvalConfig, metric: str, case_spec: dict[str, Any] | None) -> float:
|
|
360
|
+
"""The per-case threshold for ``metric``.
|
|
361
|
+
|
|
362
|
+
Precedence: the case's ``judge.<metric>.threshold`` > ``quality_metrics.<metric>.threshold``
|
|
363
|
+
> ``judges.<metric>.threshold`` > a custom metric's threshold. None of them
|
|
364
|
+
is a configuration error (a quality metric with no threshold cannot gate).
|
|
365
|
+
"""
|
|
366
|
+
if case_spec and case_spec.get("threshold") is not None:
|
|
367
|
+
return float(case_spec["threshold"])
|
|
368
|
+
quality = config.quality_metrics.get(metric)
|
|
369
|
+
if quality is not None:
|
|
370
|
+
return quality.threshold
|
|
371
|
+
judge = config.judges.get(metric)
|
|
372
|
+
if judge is not None and judge.threshold is not None:
|
|
373
|
+
return judge.threshold
|
|
374
|
+
custom = config.custom_metrics.get(metric)
|
|
375
|
+
if custom is not None:
|
|
376
|
+
return custom.threshold
|
|
377
|
+
raise EvalConfigError(
|
|
378
|
+
f"metric {metric!r} has no threshold: set judge.{metric}.threshold on the case, "
|
|
379
|
+
f"or quality_metrics.{metric}.threshold / judges.{metric}.threshold in eval_config.yaml"
|
|
380
|
+
)
|
|
381
|
+
|
|
382
|
+
|
|
383
|
+
def render_judge_prompt(
|
|
384
|
+
spec: JudgeSpec,
|
|
385
|
+
*,
|
|
386
|
+
conversation: str,
|
|
387
|
+
response: str,
|
|
388
|
+
reference: str | None,
|
|
389
|
+
context: str | None,
|
|
390
|
+
tool_calls: list[dict[str, Any]] | None,
|
|
391
|
+
transcript: str | None = None,
|
|
392
|
+
max_tool_result_chars: int | None = DEFAULT_MAX_TOOL_RESULT_CHARS,
|
|
393
|
+
) -> str:
|
|
394
|
+
"""Fill the judge prompt template; an unknown placeholder is a configuration error.
|
|
395
|
+
|
|
396
|
+
``conversation`` is everything before the reply being scored (earlier turns
|
|
397
|
+
in full, see :mod:`graph_agents_cli.eval.transcript`), ``tool_calls`` the
|
|
398
|
+
final turn's calls and ``transcript`` the whole case; without one it is
|
|
399
|
+
assembled from the other three. A tool result longer than
|
|
400
|
+
``max_tool_result_chars`` is cut with an explicit marker, never silently.
|
|
401
|
+
"""
|
|
402
|
+
reference_section = f"\nReference answer (ground truth):\n{reference}\n" if reference else ""
|
|
403
|
+
context_section = f"\nGrounding context:\n{context}\n" if context else ""
|
|
404
|
+
lines, _, _ = render_tool_calls(tool_calls, max_tool_result_chars)
|
|
405
|
+
tool_calls_section = ""
|
|
406
|
+
if lines:
|
|
407
|
+
tool_calls_section = (
|
|
408
|
+
"\nTool calls the agent made for this response:\n"
|
|
409
|
+
+ "\n".join(f"- {line}" for line in lines)
|
|
410
|
+
+ "\n"
|
|
411
|
+
)
|
|
412
|
+
if transcript is None:
|
|
413
|
+
transcript = "\n".join(
|
|
414
|
+
[conversation, *(f"agent tool call: {line}" for line in lines), f"agent: {response}"]
|
|
415
|
+
)
|
|
416
|
+
values = {
|
|
417
|
+
"metric": spec.name,
|
|
418
|
+
"rubric": spec.rubric,
|
|
419
|
+
"conversation": conversation,
|
|
420
|
+
"transcript": transcript,
|
|
421
|
+
"response": response,
|
|
422
|
+
"reference": reference or "",
|
|
423
|
+
"context": context or "",
|
|
424
|
+
"reference_section": reference_section,
|
|
425
|
+
"context_section": context_section,
|
|
426
|
+
"tool_calls_section": tool_calls_section,
|
|
427
|
+
"scale": int(spec.scale) if float(spec.scale).is_integer() else spec.scale,
|
|
428
|
+
}
|
|
429
|
+
try:
|
|
430
|
+
return spec.prompt_template.format(**values)
|
|
431
|
+
except (KeyError, IndexError, ValueError) as exc:
|
|
432
|
+
raise EvalConfigError(
|
|
433
|
+
f"judges.{spec.name}.prompt_template has an unknown placeholder: {exc} "
|
|
434
|
+
f"(allowed: {', '.join('{' + p + '}' for p in PROMPT_PLACEHOLDERS)})"
|
|
435
|
+
) from exc
|