graph-agents-cli 0.3.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- graph_agents_cli/__init__.py +26 -0
- graph_agents_cli/_api_policy.py +2145 -0
- graph_agents_cli/_approvals.py +400 -0
- graph_agents_cli/_build.py +186 -0
- graph_agents_cli/_build_info.json +7 -0
- graph_agents_cli/_chat_client.py +462 -0
- graph_agents_cli/_click.py +157 -0
- graph_agents_cli/_defaults.py +139 -0
- graph_agents_cli/_experiments.py +64 -0
- graph_agents_cli/_http.py +192 -0
- graph_agents_cli/_output.py +83 -0
- graph_agents_cli/_project.py +462 -0
- graph_agents_cli/_remote.py +220 -0
- graph_agents_cli/_response_schema.py +264 -0
- graph_agents_cli/_runner.py +319 -0
- graph_agents_cli/_skills_check.py +274 -0
- graph_agents_cli/_tools.py +189 -0
- graph_agents_cli/_trust.py +66 -0
- graph_agents_cli/api/__init__.py +15 -0
- graph_agents_cli/api/_changes.py +506 -0
- graph_agents_cli/api/_files.py +658 -0
- graph_agents_cli/api/cmd_api.py +2480 -0
- graph_agents_cli/deploy/__init__.py +15 -0
- graph_agents_cli/deploy/_config.py +171 -0
- graph_agents_cli/deploy/_image.py +128 -0
- graph_agents_cli/deploy/_kube.py +286 -0
- graph_agents_cli/deploy/_modes.py +234 -0
- graph_agents_cli/deploy/_preflight.py +370 -0
- graph_agents_cli/deploy/_values.py +168 -0
- graph_agents_cli/deploy/cmd_deploy.py +1866 -0
- graph_agents_cli/deploy/gitops.py +562 -0
- graph_agents_cli/deploy/local_load.py +273 -0
- graph_agents_cli/dev/__init__.py +13 -0
- graph_agents_cli/dev/cmd_build.py +131 -0
- graph_agents_cli/dev/cmd_install.py +78 -0
- graph_agents_cli/dev/cmd_lint.py +119 -0
- graph_agents_cli/dev/cmd_playground.py +297 -0
- graph_agents_cli/dev/policy_check.py +1287 -0
- graph_agents_cli/eval/__init__.py +22 -0
- graph_agents_cli/eval/_client.py +670 -0
- graph_agents_cli/eval/_common.py +177 -0
- graph_agents_cli/eval/_judge.py +168 -0
- graph_agents_cli/eval/_judge_runner.py +238 -0
- graph_agents_cli/eval/_paths.py +212 -0
- graph_agents_cli/eval/checks.py +581 -0
- graph_agents_cli/eval/cmd_analyze.py +278 -0
- graph_agents_cli/eval/cmd_compare.py +284 -0
- graph_agents_cli/eval/cmd_eval_group.py +80 -0
- graph_agents_cli/eval/cmd_generate.py +558 -0
- graph_agents_cli/eval/cmd_grade.py +466 -0
- graph_agents_cli/eval/cmd_metric.py +156 -0
- graph_agents_cli/eval/cmd_run.py +370 -0
- graph_agents_cli/eval/cmd_submit.py +400 -0
- graph_agents_cli/eval/config.py +435 -0
- graph_agents_cli/eval/dataset.py +350 -0
- graph_agents_cli/eval/gate.py +420 -0
- graph_agents_cli/eval/transcript.py +192 -0
- graph_agents_cli/extension/__init__.py +13 -0
- graph_agents_cli/extension/_compat.py +86 -0
- graph_agents_cli/extension/_loader.py +293 -0
- graph_agents_cli/extension/_manifest.py +135 -0
- graph_agents_cli/extension/_overrides.py +195 -0
- graph_agents_cli/extension/_paths.py +91 -0
- graph_agents_cli/extension/_refs.py +193 -0
- graph_agents_cli/extension/_resolver.py +453 -0
- graph_agents_cli/extension/_schema.py +106 -0
- graph_agents_cli/extension/_spec.py +253 -0
- graph_agents_cli/extension/_sync.py +102 -0
- graph_agents_cli/extension/_trust.py +58 -0
- graph_agents_cli/extension/cmd_extension_add.py +259 -0
- graph_agents_cli/extension/cmd_extension_group.py +57 -0
- graph_agents_cli/extension/cmd_extension_list.py +56 -0
- graph_agents_cli/extension/cmd_extension_remove.py +61 -0
- graph_agents_cli/extension/cmd_extension_update.py +195 -0
- graph_agents_cli/info/__init__.py +13 -0
- graph_agents_cli/info/cmd_info.py +222 -0
- graph_agents_cli/infra/__init__.py +15 -0
- graph_agents_cli/infra/checks.py +1169 -0
- graph_agents_cli/infra/cmd_infra.py +103 -0
- graph_agents_cli/main.py +591 -0
- graph_agents_cli/peer/__init__.py +15 -0
- graph_agents_cli/peer/_generate.py +254 -0
- graph_agents_cli/peer/cmd_peer.py +1151 -0
- graph_agents_cli/run/__init__.py +13 -0
- graph_agents_cli/run/_local_server.py +1157 -0
- graph_agents_cli/run/_signals.py +141 -0
- graph_agents_cli/run/cmd_approvals.py +530 -0
- graph_agents_cli/run/cmd_run.py +1421 -0
- graph_agents_cli/scaffold/__init__.py +19 -0
- graph_agents_cli/scaffold/agents/README.md +24 -0
- graph_agents_cli/scaffold/agents/empty_py/.template/templateconfig.yaml +22 -0
- graph_agents_cli/scaffold/agents/langgraph/.env.example +292 -0
- graph_agents_cli/scaffold/agents/langgraph/.template/templateconfig.yaml +28 -0
- graph_agents_cli/scaffold/agents/langgraph/Dockerfile +59 -0
- graph_agents_cli/scaffold/agents/langgraph/Dockerfile.langgraph-server +59 -0
- graph_agents_cli/scaffold/agents/langgraph/README.md +571 -0
- graph_agents_cli/scaffold/agents/langgraph/api-policy.yaml +60 -0
- graph_agents_cli/scaffold/agents/langgraph/app/__init__.py +20 -0
- graph_agents_cli/scaffold/agents/langgraph/app/agent.py +174 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/__init__.py +15 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/a2a.py +2162 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/a2a_client.py +1167 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/api_client.py +4220 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/approvals.py +1349 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/auth.py +1986 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/chat.py +2962 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/checkpointer.py +432 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/content.py +569 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/db.py +580 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/limits.py +203 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/metrics.py +231 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/middleware.py +361 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/model.py +611 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/playground.py +230 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/run_locks.py +459 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/structured.py +755 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/telemetry.py +681 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/threads.py +493 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/token_exchange.py +959 -0
- graph_agents_cli/scaffold/agents/langgraph/app/fast_api_app.py +770 -0
- graph_agents_cli/scaffold/agents/langgraph/app/policies/__init__.py +55 -0
- graph_agents_cli/scaffold/agents/langgraph/app/policies/custom.py +97 -0
- graph_agents_cli/scaffold/agents/langgraph/app/tools/__init__.py +46 -0
- graph_agents_cli/scaffold/agents/langgraph/app/tools/example_api.py +92 -0
- graph_agents_cli/scaffold/agents/langgraph/app/tools/weather.py +33 -0
- graph_agents_cli/scaffold/agents/langgraph/langgraph.json +14 -0
- graph_agents_cli/scaffold/agents/langgraph/pyproject.toml +78 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/conftest.py +376 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/eval/datasets/basic-dataset.json +53 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/eval/eval_config.yaml +32 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/approval_graph.py +137 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/fake_issuer.py +216 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/fake_openai.py +357 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_a2a_outcomes.py +569 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_a2a_relay.py +479 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_api_surface.py +812 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_approvals.py +1367 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_approvals_server.py +794 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_cross_actor.py +497 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_cross_actor_server.py +247 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_history_repair.py +278 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_model_apis.py +242 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_postgres.py +637 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_resilience_postgres.py +770 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_runtime_guardrails.py +854 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_server_e2e.py +340 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_server_runtime.py +989 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_structured_answers.py +584 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_structured_server.py +222 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_token_exchange_issuer.py +650 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/load_test/.results/.placeholder +0 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/load_test/README.md +22 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/load_test/conftest.py +21 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/load_test/load_test.py +81 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_a2a_client.py +824 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_a2a_scoping.py +724 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_api_client.py +1214 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_api_client_hardening.py +716 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_api_policy_rpc.py +767 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_approval_ledger.py +1536 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_fake_model.py +115 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_jwt_policy.py +991 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_limits.py +310 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_logging.py +148 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_logging_hardening.py +271 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_policy.py +378 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_resilience.py +610 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_server_auth.py +702 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_structured.py +673 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_telemetry.py +404 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_thread_listing.py +255 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_threads.py +268 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_token_exchange.py +1320 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_untrusted_content.py +393 -0
- graph_agents_cli/scaffold/agents/langgraph/uv-fastapi.lock +2084 -0
- graph_agents_cli/scaffold/agents/langgraph/uv-langgraph-server.lock +2106 -0
- graph_agents_cli/scaffold/agents/langgraph/{{cookiecutter.agent_guidance_filename}} +129 -0
- graph_agents_cli/scaffold/base_templates/_shared/graph-agents-cli-manifest.yaml +36 -0
- graph_agents_cli/scaffold/base_templates/python/.dockerignore +32 -0
- graph_agents_cli/scaffold/base_templates/python/.github/CODEOWNERS +30 -0
- graph_agents_cli/scaffold/base_templates/python/.github/agent.env +7 -0
- graph_agents_cli/scaffold/base_templates/python/.github/workflows/pr_checks.yaml +214 -0
- graph_agents_cli/scaffold/base_templates/python/.gitignore +209 -0
- graph_agents_cli/scaffold/base_templates/python/tests/unit/test_dummy.py +23 -0
- graph_agents_cli/scaffold/base_templates/python/{{cookiecutter.agent_guidance_filename}} +35 -0
- graph_agents_cli/scaffold/cmd_scaffold_group.py +49 -0
- graph_agents_cli/scaffold/commands/__init__.py +13 -0
- graph_agents_cli/scaffold/commands/create.py +1424 -0
- graph_agents_cli/scaffold/commands/enhance.py +1652 -0
- graph_agents_cli/scaffold/commands/upgrade.py +570 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/.github/agent.env +12 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/.github/workflows/promote-to-prod.yaml +371 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/.github/workflows/staging.yaml +450 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/argocd/application-dev.yaml +43 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/argocd/application-prod.yaml +41 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/argocd/application-staging.yaml +43 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/.helmignore +14 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/Chart.yaml +21 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/examples/networkpolicy.yaml +103 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/NOTES.txt +48 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/_helpers.tpl +189 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/certificate.yaml +15 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/configmap.yaml +10 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/deployment.yaml +199 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/hpa.yaml +22 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/httproute.yaml +30 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/ingress.yaml +39 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/networkpolicy.yaml +48 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/pdb.yaml +13 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/postgresql-secret.yaml +37 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/service.yaml +15 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/serviceaccount.yaml +13 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/servicemonitor.yaml +42 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values-dev.yaml +22 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values-prod.yaml +45 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values-staging.yaml +29 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values.yaml +396 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/tests/integration/test_chart.py +269 -0
- graph_agents_cli/scaffold/deployment_targets/none/README.md +5 -0
- graph_agents_cli/scaffold/deployment_targets/none/python/README.md +6 -0
- graph_agents_cli/scaffold/utils/__init__.py +13 -0
- graph_agents_cli/scaffold/utils/backup.py +212 -0
- graph_agents_cli/scaffold/utils/build_record.py +257 -0
- graph_agents_cli/scaffold/utils/cli_options.py +184 -0
- graph_agents_cli/scaffold/utils/fs.py +83 -0
- graph_agents_cli/scaffold/utils/generate_locks.py +214 -0
- graph_agents_cli/scaffold/utils/generation_metadata.py +88 -0
- graph_agents_cli/scaffold/utils/keyedit.py +768 -0
- graph_agents_cli/scaffold/utils/keymerge.py +537 -0
- graph_agents_cli/scaffold/utils/language.py +138 -0
- graph_agents_cli/scaffold/utils/lock_utils.py +94 -0
- graph_agents_cli/scaffold/utils/logging.py +77 -0
- graph_agents_cli/scaffold/utils/manifest.py +292 -0
- graph_agents_cli/scaffold/utils/merge.py +970 -0
- graph_agents_cli/scaffold/utils/merge3.py +216 -0
- graph_agents_cli/scaffold/utils/openapi_seed.py +199 -0
- graph_agents_cli/scaffold/utils/remote_template.py +376 -0
- graph_agents_cli/scaffold/utils/template.py +1352 -0
- graph_agents_cli/scaffold/utils/upgrade.py +894 -0
- graph_agents_cli/scaffold/utils/version.py +438 -0
- graph_agents_cli/secrets/__init__.py +15 -0
- graph_agents_cli/secrets/_apply.py +954 -0
- graph_agents_cli/secrets/_required.py +188 -0
- graph_agents_cli/secrets/cmd_secrets.py +211 -0
- graph_agents_cli/setup/__init__.py +13 -0
- graph_agents_cli/setup/_antigravity.py +221 -0
- graph_agents_cli/setup/cmd_auth.py +1030 -0
- graph_agents_cli/setup/cmd_dev_token.py +513 -0
- graph_agents_cli/setup/cmd_setup.py +428 -0
- graph_agents_cli/setup/cmd_update.py +140 -0
- graph_agents_cli/skills/__init__.py +13 -0
- graph_agents_cli/skills/_bundle.py +65 -0
- graph_agents_cli/skills/data/README.md +19 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/SKILL.md +357 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/references/github-settings.md +113 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/references/gitops.md +137 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/references/kubernetes.md +315 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/references/secrets.md +160 -0
- graph_agents_cli/skills/data/graph-agents-cli-eval/SKILL.md +303 -0
- graph_agents_cli/skills/data/graph-agents-cli-eval/references/dataset_schema.md +282 -0
- graph_agents_cli/skills/data/graph-agents-cli-eval/references/metrics-guide.md +143 -0
- graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/SKILL.md +659 -0
- graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/references/langchain-models.md +124 -0
- graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/references/langgraph.md +235 -0
- graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/references/template-contract.md +477 -0
- graph_agents_cli/skills/data/graph-agents-cli-observability/SKILL.md +231 -0
- graph_agents_cli/skills/data/graph-agents-cli-observability/references/langsmith.md +46 -0
- graph_agents_cli/skills/data/graph-agents-cli-observability/references/otel.md +59 -0
- graph_agents_cli/skills/data/graph-agents-cli-scaffold/SKILL.md +414 -0
- graph_agents_cli/skills/data/graph-agents-cli-scaffold/references/flags.md +134 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/SKILL.md +478 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/brainstorming.md +118 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/commands.md +419 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/extension.md +156 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/internals.md +67 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/spec-template.md +56 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/terminology.md +119 -0
- graph_agents_cli/system/__init__.py +15 -0
- graph_agents_cli/system/_apply.py +519 -0
- graph_agents_cli/system/_checks.py +1023 -0
- graph_agents_cli/system/_deploy.py +215 -0
- graph_agents_cli/system/_model.py +363 -0
- graph_agents_cli/system/_system.py +664 -0
- graph_agents_cli/system/_views.py +208 -0
- graph_agents_cli/system/cmd_system.py +423 -0
- graph_agents_cli-0.3.1.dist-info/METADATA +162 -0
- graph_agents_cli-0.3.1.dist-info/RECORD +291 -0
- graph_agents_cli-0.3.1.dist-info/WHEEL +4 -0
- graph_agents_cli-0.3.1.dist-info/entry_points.txt +2 -0
- graph_agents_cli-0.3.1.dist-info/licenses/LICENSE +201 -0
- graph_agents_cli-0.3.1.dist-info/licenses/NOTICE +19 -0
|
@@ -0,0 +1,177 @@
|
|
|
1
|
+
# Copyright 2026 graph-agents-cli contributors
|
|
2
|
+
#
|
|
3
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
4
|
+
# you may not use this file except in compliance with the License.
|
|
5
|
+
# You may obtain a copy of the License at
|
|
6
|
+
#
|
|
7
|
+
# https://www.apache.org/licenses/LICENSE-2.0
|
|
8
|
+
#
|
|
9
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
10
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
11
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
12
|
+
# See the License for the specific language governing permissions and
|
|
13
|
+
# limitations under the License.
|
|
14
|
+
|
|
15
|
+
"""Shared helpers for the eval commands: exit codes, hashing, project metadata."""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import hashlib
|
|
20
|
+
import json
|
|
21
|
+
import os
|
|
22
|
+
import tomllib
|
|
23
|
+
from datetime import UTC, datetime
|
|
24
|
+
from pathlib import Path
|
|
25
|
+
from typing import Any
|
|
26
|
+
|
|
27
|
+
import click
|
|
28
|
+
import yaml
|
|
29
|
+
|
|
30
|
+
# Exit codes of the evaluation gate (see gate.py).
|
|
31
|
+
EXIT_OK = 0
|
|
32
|
+
EXIT_GATE_FAILED = 1
|
|
33
|
+
EXIT_INCOMPLETE = 2
|
|
34
|
+
EXIT_CONFIG_ERROR = 3
|
|
35
|
+
|
|
36
|
+
MANIFEST_FILENAME = "graph-agents-cli-manifest.yaml"
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class EvalConfigError(click.ClickException):
|
|
40
|
+
"""A configuration error: unknown metric, unreachable judge, bad dataset (exit 3)."""
|
|
41
|
+
|
|
42
|
+
exit_code = EXIT_CONFIG_ERROR
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def worst_exit_code(*codes: int) -> int:
|
|
46
|
+
"""The worse of several gate exit codes (3 > 2 > 1 > 0)."""
|
|
47
|
+
return max(codes) if codes else EXIT_OK
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def utc_now_iso() -> str:
|
|
51
|
+
return datetime.now(UTC).isoformat(timespec="seconds")
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def canonical_hash(obj: Any) -> str:
|
|
55
|
+
"""sha256 of the canonical JSON encoding of ``obj`` (sorted keys, no whitespace)."""
|
|
56
|
+
encoded = json.dumps(obj, sort_keys=True, separators=(",", ":"), ensure_ascii=False)
|
|
57
|
+
return hashlib.sha256(encoded.encode("utf-8")).hexdigest()
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def load_json_file(path: Path, what: str = "file") -> Any:
|
|
61
|
+
"""Read a JSON file, raising :class:`EvalConfigError` when unreadable or invalid."""
|
|
62
|
+
try:
|
|
63
|
+
return json.loads(Path(path).read_text(encoding="utf-8"))
|
|
64
|
+
except FileNotFoundError as exc:
|
|
65
|
+
raise EvalConfigError(f"{what} not found: {path}") from exc
|
|
66
|
+
except (OSError, json.JSONDecodeError) as exc:
|
|
67
|
+
raise EvalConfigError(f"{what} is not valid JSON ({path}): {exc}") from exc
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def write_json_file(path: Path, data: Any) -> None:
|
|
71
|
+
path = Path(path)
|
|
72
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
73
|
+
path.write_text(json.dumps(data, indent=2, ensure_ascii=False) + "\n", encoding="utf-8")
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def project_env(project_root: Path | None) -> dict[str, str]:
|
|
77
|
+
"""The project's ``.env`` values overlaid with the process environment.
|
|
78
|
+
|
|
79
|
+
Only the keys present in one of the two sources are returned. The process
|
|
80
|
+
environment wins so ``JUDGE_MODEL_PROVIDER=fake graph-agents-cli eval grade``
|
|
81
|
+
behaves as expected.
|
|
82
|
+
"""
|
|
83
|
+
values: dict[str, str] = {}
|
|
84
|
+
if project_root is not None:
|
|
85
|
+
env_file = Path(project_root) / ".env"
|
|
86
|
+
if env_file.is_file():
|
|
87
|
+
from dotenv import dotenv_values
|
|
88
|
+
|
|
89
|
+
for key, value in dotenv_values(env_file).items():
|
|
90
|
+
if value is not None:
|
|
91
|
+
values[key] = value
|
|
92
|
+
for key, value in os.environ.items():
|
|
93
|
+
if key in values or key.startswith(("MODEL_", "JUDGE_", "TRACE_", "LANGSMITH_", "API_")):
|
|
94
|
+
values[key] = value
|
|
95
|
+
return values
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def read_manifest(project_root: Path | None) -> dict[str, Any]:
|
|
99
|
+
"""The raw project manifest, or ``{}`` when absent or unreadable."""
|
|
100
|
+
if project_root is None:
|
|
101
|
+
return {}
|
|
102
|
+
path = Path(project_root) / MANIFEST_FILENAME
|
|
103
|
+
if not path.is_file():
|
|
104
|
+
return {}
|
|
105
|
+
try:
|
|
106
|
+
data = yaml.safe_load(path.read_text(encoding="utf-8"))
|
|
107
|
+
except (OSError, yaml.YAMLError):
|
|
108
|
+
return {}
|
|
109
|
+
return data if isinstance(data, dict) else {}
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def project_meta(project_root: Path | None) -> dict[str, Any]:
|
|
113
|
+
"""Identity recorded in traces and results: name, agent directory, model, version.
|
|
114
|
+
|
|
115
|
+
Reads the manifest (``create_params.model_provider`` / ``model``), the
|
|
116
|
+
project's ``.env`` (``MODEL_PROVIDER`` / ``MODEL_NAME`` win when set) and
|
|
117
|
+
``pyproject.toml`` (``project.version``). Every field degrades to ``None``
|
|
118
|
+
outside a project so ``eval grade --traces ... --output ...`` works anywhere.
|
|
119
|
+
"""
|
|
120
|
+
manifest = read_manifest(project_root)
|
|
121
|
+
create_params = manifest.get("create_params") or {}
|
|
122
|
+
if not isinstance(create_params, dict):
|
|
123
|
+
create_params = {}
|
|
124
|
+
env = project_env(project_root)
|
|
125
|
+
|
|
126
|
+
provider = env.get("MODEL_PROVIDER") or create_params.get("model_provider")
|
|
127
|
+
model_name = env.get("MODEL_NAME") or create_params.get("model")
|
|
128
|
+
version = None
|
|
129
|
+
if project_root is not None:
|
|
130
|
+
pyproject = Path(project_root) / "pyproject.toml"
|
|
131
|
+
if pyproject.is_file():
|
|
132
|
+
try:
|
|
133
|
+
with open(pyproject, "rb") as f:
|
|
134
|
+
version = (tomllib.load(f).get("project") or {}).get("version")
|
|
135
|
+
except (OSError, tomllib.TOMLDecodeError):
|
|
136
|
+
version = None
|
|
137
|
+
|
|
138
|
+
return {
|
|
139
|
+
"name": manifest.get("name"),
|
|
140
|
+
"agent_directory": manifest.get("agent_directory") or "app",
|
|
141
|
+
"runtime": create_params.get("runtime"),
|
|
142
|
+
"model_provider": provider,
|
|
143
|
+
"model_name": model_name,
|
|
144
|
+
"model": f"{provider}/{model_name}"
|
|
145
|
+
if provider and model_name
|
|
146
|
+
else (model_name or provider),
|
|
147
|
+
"agent_version": str(version) if version is not None else None,
|
|
148
|
+
"capture": env.get("TRACE_CAPTURE") or "metadata",
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def resolve_judge_identity(
|
|
153
|
+
project_root: Path | None,
|
|
154
|
+
*,
|
|
155
|
+
config_provider: str | None,
|
|
156
|
+
config_model: str | None,
|
|
157
|
+
flag_provider: str | None,
|
|
158
|
+
flag_model: str | None,
|
|
159
|
+
) -> dict[str, str | None]:
|
|
160
|
+
"""The judge provider/model to record and to hand the runner.
|
|
161
|
+
|
|
162
|
+
Precedence: CLI flag > ``eval_config.yaml`` ``judge:`` > ``JUDGE_MODEL_*``
|
|
163
|
+
in ``.env`` / environment > the agent's ``MODEL_PROVIDER`` / ``MODEL_NAME``.
|
|
164
|
+
``None`` means "let ``get_judge_model()`` decide".
|
|
165
|
+
"""
|
|
166
|
+
env = project_env(project_root)
|
|
167
|
+
provider = (
|
|
168
|
+
flag_provider
|
|
169
|
+
or config_provider
|
|
170
|
+
or env.get("JUDGE_MODEL_PROVIDER")
|
|
171
|
+
or env.get("MODEL_PROVIDER")
|
|
172
|
+
or None
|
|
173
|
+
)
|
|
174
|
+
model = (
|
|
175
|
+
flag_model or config_model or env.get("JUDGE_MODEL_NAME") or env.get("MODEL_NAME") or None
|
|
176
|
+
)
|
|
177
|
+
return {"provider": provider, "model": model}
|
|
@@ -0,0 +1,168 @@
|
|
|
1
|
+
# Copyright 2026 graph-agents-cli contributors
|
|
2
|
+
#
|
|
3
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
4
|
+
# you may not use this file except in compliance with the License.
|
|
5
|
+
# You may obtain a copy of the License at
|
|
6
|
+
#
|
|
7
|
+
# https://www.apache.org/licenses/LICENSE-2.0
|
|
8
|
+
#
|
|
9
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
10
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
11
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
12
|
+
# See the License for the specific language governing permissions and
|
|
13
|
+
# limitations under the License.
|
|
14
|
+
|
|
15
|
+
"""Stage and run the judge runner inside the project's environment.
|
|
16
|
+
|
|
17
|
+
The CLI never imports LangChain. ``_judge_runner.py`` is copied to
|
|
18
|
+
``<project>/.graph-agents-cli/judge_runner.py`` and executed with
|
|
19
|
+
``uv run python .graph-agents-cli/judge_runner.py <input.json> <output.json>``
|
|
20
|
+
from the project root, so it resolves ``<agent_directory>.app_utils.model.get_judge_model()``
|
|
21
|
+
and ``langchain_core`` from the project's own lock. Prompts are rendered here;
|
|
22
|
+
the runner only invokes the model (or a custom metric callable) and parses
|
|
23
|
+
scores.
|
|
24
|
+
|
|
25
|
+
Input payload (``agent_directory`` is added from the manifest)::
|
|
26
|
+
|
|
27
|
+
{"judge": {"provider": null, "model": null}, "agent_directory": "app",
|
|
28
|
+
"items": [{"id": "greeting/response_quality", "kind": "judge",
|
|
29
|
+
"case_id": "greeting", "metric": "response_quality",
|
|
30
|
+
"prompt": "...", "scale": 5},
|
|
31
|
+
{"id": "greeting/my_metric", "kind": "custom", "case_id": "...",
|
|
32
|
+
"metric": "my_metric", "callable": "module:function",
|
|
33
|
+
"case": {...}, "trace": {...}},
|
|
34
|
+
{"id": "summary", "kind": "summarize", "prompt": "..."}]}
|
|
35
|
+
|
|
36
|
+
Output payload::
|
|
37
|
+
|
|
38
|
+
{"provider": "...", "model": "...",
|
|
39
|
+
"results": [{"id": "...", "case_id": "...", "metric": "...",
|
|
40
|
+
"score": 4, "reasoning": "...", "error": null}]}
|
|
41
|
+
"""
|
|
42
|
+
|
|
43
|
+
from __future__ import annotations
|
|
44
|
+
|
|
45
|
+
import os
|
|
46
|
+
import shutil
|
|
47
|
+
import subprocess
|
|
48
|
+
from importlib import resources
|
|
49
|
+
from pathlib import Path
|
|
50
|
+
from typing import Any
|
|
51
|
+
|
|
52
|
+
from graph_agents_cli.eval import _paths
|
|
53
|
+
from graph_agents_cli.eval._common import (
|
|
54
|
+
EvalConfigError,
|
|
55
|
+
load_json_file,
|
|
56
|
+
read_manifest,
|
|
57
|
+
write_json_file,
|
|
58
|
+
)
|
|
59
|
+
|
|
60
|
+
RUNNER_SOURCE = "_judge_runner.py"
|
|
61
|
+
RUNNER_STAGED_NAME = "judge_runner.py"
|
|
62
|
+
DEFAULT_TIMEOUT = 600 # seconds
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def stage_runner(project_root: Path) -> Path:
|
|
66
|
+
"""Copy the runner into ``<project>/.graph-agents-cli/judge_runner.py``.
|
|
67
|
+
|
|
68
|
+
Loaded through :mod:`importlib.resources` so it works from a wheel or an
|
|
69
|
+
editable install. Overwritten on every run; the file is not removed
|
|
70
|
+
afterwards (it lives in the run-state directory, which is gitignored by
|
|
71
|
+
the template).
|
|
72
|
+
"""
|
|
73
|
+
dest_dir = _paths.stage_dir(project_root)
|
|
74
|
+
dest = dest_dir / RUNNER_STAGED_NAME
|
|
75
|
+
src = resources.files("graph_agents_cli.eval").joinpath(RUNNER_SOURCE)
|
|
76
|
+
with resources.as_file(src) as src_path:
|
|
77
|
+
shutil.copyfile(src_path, dest)
|
|
78
|
+
return dest
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def run_judge_runner(
|
|
82
|
+
project_root: Path,
|
|
83
|
+
payload: dict[str, Any],
|
|
84
|
+
*,
|
|
85
|
+
timeout: int = DEFAULT_TIMEOUT,
|
|
86
|
+
) -> dict[str, Any]:
|
|
87
|
+
"""Execute the runner over ``payload`` and return its parsed output.
|
|
88
|
+
|
|
89
|
+
A runner that cannot start, cannot build the judge model, or exits
|
|
90
|
+
non-zero is an *unreachable judge*: :class:`EvalConfigError` (exit 3).
|
|
91
|
+
Per-item failures come back in ``results[*].error`` and are the caller's
|
|
92
|
+
to map onto case statuses.
|
|
93
|
+
"""
|
|
94
|
+
from graph_agents_cli._runner import run_resolved
|
|
95
|
+
|
|
96
|
+
project_root = Path(project_root)
|
|
97
|
+
# The runner imports the judge model from the project's agent package, which
|
|
98
|
+
# `create --agent-directory` names (`app` by default).
|
|
99
|
+
payload = {
|
|
100
|
+
**payload,
|
|
101
|
+
"agent_directory": read_manifest(project_root).get("agent_directory") or "app",
|
|
102
|
+
}
|
|
103
|
+
script = stage_runner(project_root)
|
|
104
|
+
stamp = _paths.timestamp()
|
|
105
|
+
stage = _paths.stage_dir(project_root)
|
|
106
|
+
input_path = stage / f"judge_input_{stamp}.json"
|
|
107
|
+
output_path = stage / f"judge_output_{stamp}.json"
|
|
108
|
+
write_json_file(input_path, payload)
|
|
109
|
+
|
|
110
|
+
rel_script = script.relative_to(project_root)
|
|
111
|
+
cmd = [
|
|
112
|
+
"uv",
|
|
113
|
+
"run",
|
|
114
|
+
"python",
|
|
115
|
+
str(rel_script),
|
|
116
|
+
str(input_path.relative_to(project_root)),
|
|
117
|
+
str(output_path.relative_to(project_root)),
|
|
118
|
+
]
|
|
119
|
+
# The template's get_judge_model() reads JUDGE_MODEL_PROVIDER / JUDGE_MODEL_NAME;
|
|
120
|
+
# the runner sets them as well, but a child that loads .env at import must
|
|
121
|
+
# already see the resolved values.
|
|
122
|
+
env = dict(os.environ)
|
|
123
|
+
judge = payload.get("judge") or {}
|
|
124
|
+
if judge.get("provider"):
|
|
125
|
+
env["JUDGE_MODEL_PROVIDER"] = str(judge["provider"])
|
|
126
|
+
if judge.get("model"):
|
|
127
|
+
env["JUDGE_MODEL_NAME"] = str(judge["model"])
|
|
128
|
+
try:
|
|
129
|
+
result = run_resolved(
|
|
130
|
+
cmd,
|
|
131
|
+
cwd=str(project_root),
|
|
132
|
+
capture_output=True,
|
|
133
|
+
text=True,
|
|
134
|
+
encoding="utf-8",
|
|
135
|
+
errors="replace",
|
|
136
|
+
timeout=timeout,
|
|
137
|
+
env=env,
|
|
138
|
+
)
|
|
139
|
+
except subprocess.TimeoutExpired as exc:
|
|
140
|
+
raise EvalConfigError(
|
|
141
|
+
f"judge runner timed out after {timeout}s; the judge model may be unreachable"
|
|
142
|
+
) from exc
|
|
143
|
+
except Exception as exc: # ToolNotFoundError (uv missing), OSError
|
|
144
|
+
raise EvalConfigError(f"judge runner could not start: {exc}") from exc
|
|
145
|
+
|
|
146
|
+
if result.returncode != 0:
|
|
147
|
+
detail = "\n".join(
|
|
148
|
+
part.strip() for part in (result.stdout, result.stderr) if part and part.strip()
|
|
149
|
+
)
|
|
150
|
+
raise EvalConfigError(
|
|
151
|
+
f"judge runner failed (exit code {result.returncode}); the judge model is "
|
|
152
|
+
f"unreachable or misconfigured.\n{detail}".rstrip()
|
|
153
|
+
)
|
|
154
|
+
if not output_path.exists():
|
|
155
|
+
raise EvalConfigError("judge runner exited 0 but wrote no output file")
|
|
156
|
+
output = load_json_file(output_path, "judge runner output")
|
|
157
|
+
if not isinstance(output, dict) or not isinstance(output.get("results"), list):
|
|
158
|
+
raise EvalConfigError("judge runner output is malformed (no 'results' list)")
|
|
159
|
+
try:
|
|
160
|
+
input_path.unlink()
|
|
161
|
+
output_path.unlink()
|
|
162
|
+
except OSError:
|
|
163
|
+
pass
|
|
164
|
+
return output
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def results_by_id(output: dict[str, Any]) -> dict[str, dict[str, Any]]:
|
|
168
|
+
return {str(r.get("id")): r for r in output.get("results", []) if isinstance(r, dict)}
|
|
@@ -0,0 +1,238 @@
|
|
|
1
|
+
# Copyright 2026 graph-agents-cli contributors
|
|
2
|
+
#
|
|
3
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
4
|
+
# you may not use this file except in compliance with the License.
|
|
5
|
+
# You may obtain a copy of the License at
|
|
6
|
+
#
|
|
7
|
+
# https://www.apache.org/licenses/LICENSE-2.0
|
|
8
|
+
#
|
|
9
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
10
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
11
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
12
|
+
# See the License for the specific language governing permissions and
|
|
13
|
+
# limitations under the License.
|
|
14
|
+
|
|
15
|
+
"""Judge runner for ``graph-agents-cli eval grade`` and ``eval analyze --judge``.
|
|
16
|
+
|
|
17
|
+
This file is staged into the user's project as ``.graph-agents-cli/judge_runner.py``
|
|
18
|
+
and executed there with ``uv run python .graph-agents-cli/judge_runner.py
|
|
19
|
+
<input.json> <output.json>`` so it runs inside the project's environment.
|
|
20
|
+
Do not edit the staged copy; it is overwritten on every run.
|
|
21
|
+
|
|
22
|
+
It imports ``<agent_directory>.app_utils.model.get_judge_model()`` (the
|
|
23
|
+
payload's ``agent_directory``, the project's agent package, ``app`` by default;
|
|
24
|
+
the template contract: no arguments, judge chosen from ``JUDGE_MODEL_PROVIDER`` /
|
|
25
|
+
``JUDGE_MODEL_NAME``)
|
|
26
|
+
and ``langchain_core`` lazily; under the ``fake`` provider it returns a
|
|
27
|
+
deterministic score equal to the scale so tests and CI run without keys.
|
|
28
|
+
It never imports graph_agents_cli.
|
|
29
|
+
"""
|
|
30
|
+
|
|
31
|
+
from __future__ import annotations
|
|
32
|
+
|
|
33
|
+
import importlib
|
|
34
|
+
import json
|
|
35
|
+
import os
|
|
36
|
+
import re
|
|
37
|
+
import sys
|
|
38
|
+
import traceback
|
|
39
|
+
from pathlib import Path
|
|
40
|
+
from typing import Any
|
|
41
|
+
|
|
42
|
+
FAKE_PROVIDER = "fake"
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _load_dotenv() -> None:
|
|
46
|
+
try:
|
|
47
|
+
from dotenv import load_dotenv # the template depends on python-dotenv
|
|
48
|
+
except ImportError:
|
|
49
|
+
return
|
|
50
|
+
load_dotenv(Path.cwd() / ".env")
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def resolve_provider(payload: dict[str, Any]) -> str | None:
|
|
54
|
+
judge = payload.get("judge") or {}
|
|
55
|
+
return (
|
|
56
|
+
judge.get("provider")
|
|
57
|
+
or os.environ.get("JUDGE_MODEL_PROVIDER")
|
|
58
|
+
or os.environ.get("MODEL_PROVIDER")
|
|
59
|
+
or None
|
|
60
|
+
)
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def is_fake(provider: str | None) -> bool:
|
|
64
|
+
return (provider or "").lower() == FAKE_PROVIDER
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
# A Python package name: what `create --agent-directory` accepts.
|
|
68
|
+
_PACKAGE = re.compile(r"^[A-Za-z_][A-Za-z0-9_]*$")
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def agent_package(payload: dict[str, Any]) -> str:
|
|
72
|
+
"""The project's agent package (the manifest's ``agent_directory``), ``app`` by default."""
|
|
73
|
+
name = payload.get("agent_directory")
|
|
74
|
+
return name if isinstance(name, str) and _PACKAGE.fullmatch(name) else "app"
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def build_model(provider: str | None, model: str | None, package: str = "app") -> Any:
|
|
78
|
+
"""``<package>.app_utils.model.get_judge_model()`` from the project.
|
|
79
|
+
|
|
80
|
+
The template resolves the judge from ``JUDGE_MODEL_PROVIDER`` /
|
|
81
|
+
``JUDGE_MODEL_NAME`` (falling back to ``MODEL_*``) when called, so the
|
|
82
|
+
overrides the CLI resolved travel through the environment, set after the
|
|
83
|
+
import so an import-time ``.env`` load cannot shadow them.
|
|
84
|
+
"""
|
|
85
|
+
if str(Path.cwd()) not in sys.path:
|
|
86
|
+
sys.path.insert(0, str(Path.cwd()))
|
|
87
|
+
module = importlib.import_module(f"{package}.app_utils.model")
|
|
88
|
+
if provider:
|
|
89
|
+
os.environ["JUDGE_MODEL_PROVIDER"] = provider
|
|
90
|
+
if model:
|
|
91
|
+
os.environ["JUDGE_MODEL_NAME"] = model
|
|
92
|
+
return module.get_judge_model()
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def invoke_model(model: Any, prompt: str) -> str:
|
|
96
|
+
"""Send ``prompt`` to a LangChain chat model and return the text of its reply."""
|
|
97
|
+
try:
|
|
98
|
+
from langchain_core.messages import HumanMessage
|
|
99
|
+
|
|
100
|
+
result = model.invoke([HumanMessage(content=prompt)])
|
|
101
|
+
except ImportError:
|
|
102
|
+
result = model.invoke(prompt)
|
|
103
|
+
content = getattr(result, "content", result)
|
|
104
|
+
if isinstance(content, list):
|
|
105
|
+
parts = []
|
|
106
|
+
for part in content:
|
|
107
|
+
if isinstance(part, dict):
|
|
108
|
+
parts.append(str(part.get("text", "")))
|
|
109
|
+
else:
|
|
110
|
+
parts.append(str(part))
|
|
111
|
+
content = "".join(parts)
|
|
112
|
+
return str(content)
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
_JSON_OBJECT = re.compile(r"\{.*\}", re.DOTALL)
|
|
116
|
+
_SCORE = re.compile(r"score\W{0,5}(-?\d+(?:\.\d+)?)", re.IGNORECASE)
|
|
117
|
+
_NUMBER = re.compile(r"-?\d+(?:\.\d+)?")
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def parse_score(text: str) -> tuple[float, str]:
|
|
121
|
+
"""Extract ``(score, reasoning)`` from a judge reply.
|
|
122
|
+
|
|
123
|
+
Accepts a JSON object (optionally fenced or embedded in prose) with a
|
|
124
|
+
``score`` key, a ``score: N`` phrase, or, as a last resort, a bare number.
|
|
125
|
+
Raises ``ValueError`` when no score can be read.
|
|
126
|
+
"""
|
|
127
|
+
stripped = text.strip()
|
|
128
|
+
candidates = [stripped]
|
|
129
|
+
match = _JSON_OBJECT.search(stripped)
|
|
130
|
+
if match:
|
|
131
|
+
candidates.append(match.group(0))
|
|
132
|
+
for candidate in candidates:
|
|
133
|
+
try:
|
|
134
|
+
data = json.loads(candidate)
|
|
135
|
+
except (json.JSONDecodeError, ValueError):
|
|
136
|
+
continue
|
|
137
|
+
if isinstance(data, dict) and "score" in data:
|
|
138
|
+
score = float(data["score"])
|
|
139
|
+
reasoning = data.get("reasoning") or data.get("explanation") or ""
|
|
140
|
+
return score, str(reasoning)
|
|
141
|
+
if isinstance(data, int | float) and not isinstance(data, bool):
|
|
142
|
+
return float(data), ""
|
|
143
|
+
match = _SCORE.search(stripped)
|
|
144
|
+
if match:
|
|
145
|
+
return float(match.group(1)), stripped
|
|
146
|
+
match = _NUMBER.fullmatch(stripped)
|
|
147
|
+
if match:
|
|
148
|
+
return float(match.group(0)), ""
|
|
149
|
+
raise ValueError(f"could not parse a score from judge reply: {stripped[:200]!r}")
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def run_custom(item: dict[str, Any]) -> tuple[float, str]:
|
|
153
|
+
"""Import ``module:function`` from the project and call it with ``(case, trace)``."""
|
|
154
|
+
target = item["callable"]
|
|
155
|
+
module_name, _, func_name = target.partition(":")
|
|
156
|
+
if str(Path.cwd()) not in sys.path:
|
|
157
|
+
sys.path.insert(0, str(Path.cwd()))
|
|
158
|
+
module = importlib.import_module(module_name)
|
|
159
|
+
func = getattr(module, func_name)
|
|
160
|
+
result = func(item.get("case") or {}, item.get("trace") or {})
|
|
161
|
+
if isinstance(result, bool):
|
|
162
|
+
return (1.0 if result else 0.0), ""
|
|
163
|
+
if isinstance(result, int | float):
|
|
164
|
+
return float(result), ""
|
|
165
|
+
if isinstance(result, dict) and "score" in result:
|
|
166
|
+
reasoning = result.get("reasoning") or result.get("explanation") or ""
|
|
167
|
+
score = result["score"]
|
|
168
|
+
if isinstance(score, bool):
|
|
169
|
+
score = 1.0 if score else 0.0
|
|
170
|
+
return float(score), str(reasoning)
|
|
171
|
+
raise ValueError(
|
|
172
|
+
f"{target} returned {type(result).__name__}; expected bool, number, or {{'score': ...}}"
|
|
173
|
+
)
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def run_items(payload: dict[str, Any], model: Any = None, *, fake: bool = False) -> dict[str, Any]:
|
|
177
|
+
"""Score every item; per-item failures become ``error`` entries, never exceptions."""
|
|
178
|
+
results: list[dict[str, Any]] = []
|
|
179
|
+
for item in payload.get("items", []):
|
|
180
|
+
entry: dict[str, Any] = {
|
|
181
|
+
"id": item.get("id"),
|
|
182
|
+
"case_id": item.get("case_id"),
|
|
183
|
+
"metric": item.get("metric"),
|
|
184
|
+
"score": None,
|
|
185
|
+
"reasoning": "",
|
|
186
|
+
"error": None,
|
|
187
|
+
}
|
|
188
|
+
kind = item.get("kind", "judge")
|
|
189
|
+
try:
|
|
190
|
+
if kind == "custom":
|
|
191
|
+
entry["score"], entry["reasoning"] = run_custom(item)
|
|
192
|
+
elif kind == "summarize":
|
|
193
|
+
if fake:
|
|
194
|
+
entry["reasoning"] = "fake provider: no summary generated"
|
|
195
|
+
else:
|
|
196
|
+
entry["reasoning"] = invoke_model(model, item["prompt"])
|
|
197
|
+
else:
|
|
198
|
+
scale = float(item.get("scale") or 5)
|
|
199
|
+
if fake:
|
|
200
|
+
entry["score"] = scale
|
|
201
|
+
entry["reasoning"] = "fake provider: deterministic score"
|
|
202
|
+
else:
|
|
203
|
+
text = invoke_model(model, item["prompt"])
|
|
204
|
+
entry["score"], entry["reasoning"] = parse_score(text)
|
|
205
|
+
except Exception as exc:
|
|
206
|
+
entry["error"] = f"{type(exc).__name__}: {exc}"
|
|
207
|
+
results.append(entry)
|
|
208
|
+
return {"results": results}
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
def main(argv: list[str]) -> int:
|
|
212
|
+
if len(argv) != 3:
|
|
213
|
+
print("usage: judge_runner.py <input.json> <output.json>", file=sys.stderr)
|
|
214
|
+
return 2
|
|
215
|
+
input_path, output_path = Path(argv[1]), Path(argv[2])
|
|
216
|
+
payload = json.loads(input_path.read_text(encoding="utf-8"))
|
|
217
|
+
_load_dotenv()
|
|
218
|
+
provider = resolve_provider(payload)
|
|
219
|
+
model_name = (payload.get("judge") or {}).get("model") or os.environ.get("JUDGE_MODEL_NAME")
|
|
220
|
+
fake = is_fake(provider)
|
|
221
|
+
needs_model = any(i.get("kind", "judge") != "custom" for i in payload.get("items", []))
|
|
222
|
+
model = None
|
|
223
|
+
if needs_model and not fake:
|
|
224
|
+
try:
|
|
225
|
+
model = build_model(provider, model_name, agent_package(payload))
|
|
226
|
+
except Exception:
|
|
227
|
+
print("judge runner: could not build the judge model", file=sys.stderr)
|
|
228
|
+
traceback.print_exc()
|
|
229
|
+
return 1
|
|
230
|
+
output = run_items(payload, model, fake=fake)
|
|
231
|
+
output["provider"] = provider
|
|
232
|
+
output["model"] = model_name or (getattr(model, "model_name", None) if model else None)
|
|
233
|
+
output_path.write_text(json.dumps(output, indent=2), encoding="utf-8")
|
|
234
|
+
return 0
|
|
235
|
+
|
|
236
|
+
|
|
237
|
+
if __name__ == "__main__":
|
|
238
|
+
sys.exit(main(sys.argv))
|