graph-agents-cli 0.3.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- graph_agents_cli/__init__.py +26 -0
- graph_agents_cli/_api_policy.py +2145 -0
- graph_agents_cli/_approvals.py +400 -0
- graph_agents_cli/_build.py +186 -0
- graph_agents_cli/_build_info.json +7 -0
- graph_agents_cli/_chat_client.py +462 -0
- graph_agents_cli/_click.py +157 -0
- graph_agents_cli/_defaults.py +139 -0
- graph_agents_cli/_experiments.py +64 -0
- graph_agents_cli/_http.py +192 -0
- graph_agents_cli/_output.py +83 -0
- graph_agents_cli/_project.py +462 -0
- graph_agents_cli/_remote.py +220 -0
- graph_agents_cli/_response_schema.py +264 -0
- graph_agents_cli/_runner.py +319 -0
- graph_agents_cli/_skills_check.py +274 -0
- graph_agents_cli/_tools.py +189 -0
- graph_agents_cli/_trust.py +66 -0
- graph_agents_cli/api/__init__.py +15 -0
- graph_agents_cli/api/_changes.py +506 -0
- graph_agents_cli/api/_files.py +658 -0
- graph_agents_cli/api/cmd_api.py +2480 -0
- graph_agents_cli/deploy/__init__.py +15 -0
- graph_agents_cli/deploy/_config.py +171 -0
- graph_agents_cli/deploy/_image.py +128 -0
- graph_agents_cli/deploy/_kube.py +286 -0
- graph_agents_cli/deploy/_modes.py +234 -0
- graph_agents_cli/deploy/_preflight.py +370 -0
- graph_agents_cli/deploy/_values.py +168 -0
- graph_agents_cli/deploy/cmd_deploy.py +1866 -0
- graph_agents_cli/deploy/gitops.py +562 -0
- graph_agents_cli/deploy/local_load.py +273 -0
- graph_agents_cli/dev/__init__.py +13 -0
- graph_agents_cli/dev/cmd_build.py +131 -0
- graph_agents_cli/dev/cmd_install.py +78 -0
- graph_agents_cli/dev/cmd_lint.py +119 -0
- graph_agents_cli/dev/cmd_playground.py +297 -0
- graph_agents_cli/dev/policy_check.py +1287 -0
- graph_agents_cli/eval/__init__.py +22 -0
- graph_agents_cli/eval/_client.py +670 -0
- graph_agents_cli/eval/_common.py +177 -0
- graph_agents_cli/eval/_judge.py +168 -0
- graph_agents_cli/eval/_judge_runner.py +238 -0
- graph_agents_cli/eval/_paths.py +212 -0
- graph_agents_cli/eval/checks.py +581 -0
- graph_agents_cli/eval/cmd_analyze.py +278 -0
- graph_agents_cli/eval/cmd_compare.py +284 -0
- graph_agents_cli/eval/cmd_eval_group.py +80 -0
- graph_agents_cli/eval/cmd_generate.py +558 -0
- graph_agents_cli/eval/cmd_grade.py +466 -0
- graph_agents_cli/eval/cmd_metric.py +156 -0
- graph_agents_cli/eval/cmd_run.py +370 -0
- graph_agents_cli/eval/cmd_submit.py +400 -0
- graph_agents_cli/eval/config.py +435 -0
- graph_agents_cli/eval/dataset.py +350 -0
- graph_agents_cli/eval/gate.py +420 -0
- graph_agents_cli/eval/transcript.py +192 -0
- graph_agents_cli/extension/__init__.py +13 -0
- graph_agents_cli/extension/_compat.py +86 -0
- graph_agents_cli/extension/_loader.py +293 -0
- graph_agents_cli/extension/_manifest.py +135 -0
- graph_agents_cli/extension/_overrides.py +195 -0
- graph_agents_cli/extension/_paths.py +91 -0
- graph_agents_cli/extension/_refs.py +193 -0
- graph_agents_cli/extension/_resolver.py +453 -0
- graph_agents_cli/extension/_schema.py +106 -0
- graph_agents_cli/extension/_spec.py +253 -0
- graph_agents_cli/extension/_sync.py +102 -0
- graph_agents_cli/extension/_trust.py +58 -0
- graph_agents_cli/extension/cmd_extension_add.py +259 -0
- graph_agents_cli/extension/cmd_extension_group.py +57 -0
- graph_agents_cli/extension/cmd_extension_list.py +56 -0
- graph_agents_cli/extension/cmd_extension_remove.py +61 -0
- graph_agents_cli/extension/cmd_extension_update.py +195 -0
- graph_agents_cli/info/__init__.py +13 -0
- graph_agents_cli/info/cmd_info.py +222 -0
- graph_agents_cli/infra/__init__.py +15 -0
- graph_agents_cli/infra/checks.py +1169 -0
- graph_agents_cli/infra/cmd_infra.py +103 -0
- graph_agents_cli/main.py +591 -0
- graph_agents_cli/peer/__init__.py +15 -0
- graph_agents_cli/peer/_generate.py +254 -0
- graph_agents_cli/peer/cmd_peer.py +1151 -0
- graph_agents_cli/run/__init__.py +13 -0
- graph_agents_cli/run/_local_server.py +1157 -0
- graph_agents_cli/run/_signals.py +141 -0
- graph_agents_cli/run/cmd_approvals.py +530 -0
- graph_agents_cli/run/cmd_run.py +1421 -0
- graph_agents_cli/scaffold/__init__.py +19 -0
- graph_agents_cli/scaffold/agents/README.md +24 -0
- graph_agents_cli/scaffold/agents/empty_py/.template/templateconfig.yaml +22 -0
- graph_agents_cli/scaffold/agents/langgraph/.env.example +292 -0
- graph_agents_cli/scaffold/agents/langgraph/.template/templateconfig.yaml +28 -0
- graph_agents_cli/scaffold/agents/langgraph/Dockerfile +59 -0
- graph_agents_cli/scaffold/agents/langgraph/Dockerfile.langgraph-server +59 -0
- graph_agents_cli/scaffold/agents/langgraph/README.md +571 -0
- graph_agents_cli/scaffold/agents/langgraph/api-policy.yaml +60 -0
- graph_agents_cli/scaffold/agents/langgraph/app/__init__.py +20 -0
- graph_agents_cli/scaffold/agents/langgraph/app/agent.py +174 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/__init__.py +15 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/a2a.py +2162 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/a2a_client.py +1167 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/api_client.py +4220 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/approvals.py +1349 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/auth.py +1986 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/chat.py +2962 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/checkpointer.py +432 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/content.py +569 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/db.py +580 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/limits.py +203 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/metrics.py +231 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/middleware.py +361 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/model.py +611 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/playground.py +230 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/run_locks.py +459 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/structured.py +755 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/telemetry.py +681 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/threads.py +493 -0
- graph_agents_cli/scaffold/agents/langgraph/app/app_utils/token_exchange.py +959 -0
- graph_agents_cli/scaffold/agents/langgraph/app/fast_api_app.py +770 -0
- graph_agents_cli/scaffold/agents/langgraph/app/policies/__init__.py +55 -0
- graph_agents_cli/scaffold/agents/langgraph/app/policies/custom.py +97 -0
- graph_agents_cli/scaffold/agents/langgraph/app/tools/__init__.py +46 -0
- graph_agents_cli/scaffold/agents/langgraph/app/tools/example_api.py +92 -0
- graph_agents_cli/scaffold/agents/langgraph/app/tools/weather.py +33 -0
- graph_agents_cli/scaffold/agents/langgraph/langgraph.json +14 -0
- graph_agents_cli/scaffold/agents/langgraph/pyproject.toml +78 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/conftest.py +376 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/eval/datasets/basic-dataset.json +53 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/eval/eval_config.yaml +32 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/approval_graph.py +137 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/fake_issuer.py +216 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/fake_openai.py +357 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_a2a_outcomes.py +569 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_a2a_relay.py +479 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_api_surface.py +812 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_approvals.py +1367 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_approvals_server.py +794 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_cross_actor.py +497 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_cross_actor_server.py +247 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_history_repair.py +278 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_model_apis.py +242 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_postgres.py +637 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_resilience_postgres.py +770 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_runtime_guardrails.py +854 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_server_e2e.py +340 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_server_runtime.py +989 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_structured_answers.py +584 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_structured_server.py +222 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/integration/test_token_exchange_issuer.py +650 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/load_test/.results/.placeholder +0 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/load_test/README.md +22 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/load_test/conftest.py +21 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/load_test/load_test.py +81 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_a2a_client.py +824 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_a2a_scoping.py +724 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_api_client.py +1214 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_api_client_hardening.py +716 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_api_policy_rpc.py +767 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_approval_ledger.py +1536 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_fake_model.py +115 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_jwt_policy.py +991 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_limits.py +310 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_logging.py +148 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_logging_hardening.py +271 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_policy.py +378 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_resilience.py +610 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_server_auth.py +702 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_structured.py +673 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_telemetry.py +404 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_thread_listing.py +255 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_threads.py +268 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_token_exchange.py +1320 -0
- graph_agents_cli/scaffold/agents/langgraph/tests/unit/test_untrusted_content.py +393 -0
- graph_agents_cli/scaffold/agents/langgraph/uv-fastapi.lock +2084 -0
- graph_agents_cli/scaffold/agents/langgraph/uv-langgraph-server.lock +2106 -0
- graph_agents_cli/scaffold/agents/langgraph/{{cookiecutter.agent_guidance_filename}} +129 -0
- graph_agents_cli/scaffold/base_templates/_shared/graph-agents-cli-manifest.yaml +36 -0
- graph_agents_cli/scaffold/base_templates/python/.dockerignore +32 -0
- graph_agents_cli/scaffold/base_templates/python/.github/CODEOWNERS +30 -0
- graph_agents_cli/scaffold/base_templates/python/.github/agent.env +7 -0
- graph_agents_cli/scaffold/base_templates/python/.github/workflows/pr_checks.yaml +214 -0
- graph_agents_cli/scaffold/base_templates/python/.gitignore +209 -0
- graph_agents_cli/scaffold/base_templates/python/tests/unit/test_dummy.py +23 -0
- graph_agents_cli/scaffold/base_templates/python/{{cookiecutter.agent_guidance_filename}} +35 -0
- graph_agents_cli/scaffold/cmd_scaffold_group.py +49 -0
- graph_agents_cli/scaffold/commands/__init__.py +13 -0
- graph_agents_cli/scaffold/commands/create.py +1424 -0
- graph_agents_cli/scaffold/commands/enhance.py +1652 -0
- graph_agents_cli/scaffold/commands/upgrade.py +570 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/.github/agent.env +12 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/.github/workflows/promote-to-prod.yaml +371 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/.github/workflows/staging.yaml +450 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/argocd/application-dev.yaml +43 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/argocd/application-prod.yaml +41 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/argocd/application-staging.yaml +43 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/.helmignore +14 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/Chart.yaml +21 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/examples/networkpolicy.yaml +103 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/NOTES.txt +48 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/_helpers.tpl +189 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/certificate.yaml +15 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/configmap.yaml +10 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/deployment.yaml +199 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/hpa.yaml +22 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/httproute.yaml +30 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/ingress.yaml +39 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/networkpolicy.yaml +48 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/pdb.yaml +13 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/postgresql-secret.yaml +37 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/service.yaml +15 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/serviceaccount.yaml +13 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/templates/servicemonitor.yaml +42 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values-dev.yaml +22 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values-prod.yaml +45 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values-staging.yaml +29 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/deployment/helm/{{cookiecutter.project_name}}/values.yaml +396 -0
- graph_agents_cli/scaffold/deployment_targets/kubernetes/python/tests/integration/test_chart.py +269 -0
- graph_agents_cli/scaffold/deployment_targets/none/README.md +5 -0
- graph_agents_cli/scaffold/deployment_targets/none/python/README.md +6 -0
- graph_agents_cli/scaffold/utils/__init__.py +13 -0
- graph_agents_cli/scaffold/utils/backup.py +212 -0
- graph_agents_cli/scaffold/utils/build_record.py +257 -0
- graph_agents_cli/scaffold/utils/cli_options.py +184 -0
- graph_agents_cli/scaffold/utils/fs.py +83 -0
- graph_agents_cli/scaffold/utils/generate_locks.py +214 -0
- graph_agents_cli/scaffold/utils/generation_metadata.py +88 -0
- graph_agents_cli/scaffold/utils/keyedit.py +768 -0
- graph_agents_cli/scaffold/utils/keymerge.py +537 -0
- graph_agents_cli/scaffold/utils/language.py +138 -0
- graph_agents_cli/scaffold/utils/lock_utils.py +94 -0
- graph_agents_cli/scaffold/utils/logging.py +77 -0
- graph_agents_cli/scaffold/utils/manifest.py +292 -0
- graph_agents_cli/scaffold/utils/merge.py +970 -0
- graph_agents_cli/scaffold/utils/merge3.py +216 -0
- graph_agents_cli/scaffold/utils/openapi_seed.py +199 -0
- graph_agents_cli/scaffold/utils/remote_template.py +376 -0
- graph_agents_cli/scaffold/utils/template.py +1352 -0
- graph_agents_cli/scaffold/utils/upgrade.py +894 -0
- graph_agents_cli/scaffold/utils/version.py +438 -0
- graph_agents_cli/secrets/__init__.py +15 -0
- graph_agents_cli/secrets/_apply.py +954 -0
- graph_agents_cli/secrets/_required.py +188 -0
- graph_agents_cli/secrets/cmd_secrets.py +211 -0
- graph_agents_cli/setup/__init__.py +13 -0
- graph_agents_cli/setup/_antigravity.py +221 -0
- graph_agents_cli/setup/cmd_auth.py +1030 -0
- graph_agents_cli/setup/cmd_dev_token.py +513 -0
- graph_agents_cli/setup/cmd_setup.py +428 -0
- graph_agents_cli/setup/cmd_update.py +140 -0
- graph_agents_cli/skills/__init__.py +13 -0
- graph_agents_cli/skills/_bundle.py +65 -0
- graph_agents_cli/skills/data/README.md +19 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/SKILL.md +357 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/references/github-settings.md +113 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/references/gitops.md +137 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/references/kubernetes.md +315 -0
- graph_agents_cli/skills/data/graph-agents-cli-deploy/references/secrets.md +160 -0
- graph_agents_cli/skills/data/graph-agents-cli-eval/SKILL.md +303 -0
- graph_agents_cli/skills/data/graph-agents-cli-eval/references/dataset_schema.md +282 -0
- graph_agents_cli/skills/data/graph-agents-cli-eval/references/metrics-guide.md +143 -0
- graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/SKILL.md +659 -0
- graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/references/langchain-models.md +124 -0
- graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/references/langgraph.md +235 -0
- graph_agents_cli/skills/data/graph-agents-cli-langgraph-code/references/template-contract.md +477 -0
- graph_agents_cli/skills/data/graph-agents-cli-observability/SKILL.md +231 -0
- graph_agents_cli/skills/data/graph-agents-cli-observability/references/langsmith.md +46 -0
- graph_agents_cli/skills/data/graph-agents-cli-observability/references/otel.md +59 -0
- graph_agents_cli/skills/data/graph-agents-cli-scaffold/SKILL.md +414 -0
- graph_agents_cli/skills/data/graph-agents-cli-scaffold/references/flags.md +134 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/SKILL.md +478 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/brainstorming.md +118 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/commands.md +419 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/extension.md +156 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/internals.md +67 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/spec-template.md +56 -0
- graph_agents_cli/skills/data/graph-agents-cli-workflow/references/terminology.md +119 -0
- graph_agents_cli/system/__init__.py +15 -0
- graph_agents_cli/system/_apply.py +519 -0
- graph_agents_cli/system/_checks.py +1023 -0
- graph_agents_cli/system/_deploy.py +215 -0
- graph_agents_cli/system/_model.py +363 -0
- graph_agents_cli/system/_system.py +664 -0
- graph_agents_cli/system/_views.py +208 -0
- graph_agents_cli/system/cmd_system.py +423 -0
- graph_agents_cli-0.3.1.dist-info/METADATA +162 -0
- graph_agents_cli-0.3.1.dist-info/RECORD +291 -0
- graph_agents_cli-0.3.1.dist-info/WHEEL +4 -0
- graph_agents_cli-0.3.1.dist-info/entry_points.txt +2 -0
- graph_agents_cli-0.3.1.dist-info/licenses/LICENSE +201 -0
- graph_agents_cli-0.3.1.dist-info/licenses/NOTICE +19 -0
|
@@ -0,0 +1,143 @@
|
|
|
1
|
+
# Metrics guide
|
|
2
|
+
|
|
3
|
+
Two kinds of checks: **deterministic `expect` checks** (in-process, no model, always mandatory)
|
|
4
|
+
and **judge metrics** (a chat model scores the response against a rubric; mandatory unless
|
|
5
|
+
designated a quality metric). `graph-agents-cli eval metric list` prints the live set.
|
|
6
|
+
|
|
7
|
+
## Deterministic checks (`expect`)
|
|
8
|
+
|
|
9
|
+
| Check | Passes when | Typical use |
|
|
10
|
+
|---|---|---|
|
|
11
|
+
| `contains: [..]` | every substring occurs in the final response (case-insensitive) | key facts, required disclaimers |
|
|
12
|
+
| `not_contains: [..]` | none occurs (case-insensitive) | leaked secrets, forbidden claims, refusal text when a call should have succeeded |
|
|
13
|
+
| `regex: "..."` | `re.search` matches (`(?i)` to ignore case) | ids, dates, numeric formats |
|
|
14
|
+
| `json_schema: {..}` | the final answer validates against the schema: the run's `structured_response` as it is when the project has a response schema, else the final reply's JSON (the whole reply, else its last JSON object or array of the schema's root type) | structured answers; case-specific values (`const`, `enum`) on top of the project's own schema |
|
|
15
|
+
| `tool_calls: [{name, args_subset}]` | each entry matches a trace tool call by name with `args_subset` a subset of the recorded args; `ordered: true` enforces relative order | trajectory assertions, policy regressions (`getIncident` called, nothing else) |
|
|
16
|
+
| `no_tool_calls: true` | the trace has no tool call | answers that must come from the prompt alone |
|
|
17
|
+
| `max_latency_ms: n` | `latency_ms <= n` | latency budget |
|
|
18
|
+
| `max_tokens: n` | `input_tokens + output_tokens <= n` | cost budget |
|
|
19
|
+
| `approvals: [{match, status}]` | each entry matches a distinct gate the run hit (trace `approvals`): `match` by `operation_id` and/or `method` + `path` (template), optional `api`; `status` `gated` (default, any outcome), `approved` or `rejected` | writes that must wait for a human, and how the case decided them |
|
|
20
|
+
| `no_approvals: true` | no call reached an approval gate | injection cases: the planted write never got as far as asking |
|
|
21
|
+
|
|
22
|
+
All present checks must pass; a failed check marks the case `failed` with a reason.
|
|
23
|
+
|
|
24
|
+
Modifiers (keys of `expect` that are not checks; `eval metric list` shows them too):
|
|
25
|
+
|
|
26
|
+
| Modifier | Default | Effect |
|
|
27
|
+
|---|---|---|
|
|
28
|
+
| `ordered` | `false` | `tool_calls` must appear in the listed order |
|
|
29
|
+
| `case_insensitive` | `true` | `contains` / `not_contains` ignore case (casefold); `false` for exact case |
|
|
30
|
+
| `scope` | `final_turn` | on a multi-turn case, `final_turn` reads the final reply and its tool calls, latency and usage; `all_turns` reads every turn: `contains`/`regex` pass when any reply matches, `not_contains` fails when any does, `tool_calls`/`no_tool_calls` read every call in order, `approvals`/`no_approvals` every gate, `max_latency_ms` bounds each turn, `max_tokens` their sum. `json_schema` always reads the final answer |
|
|
31
|
+
|
|
32
|
+
Case-insensitive matching is the safer default for `not_contains`: a refusal check such as
|
|
33
|
+
`not_contains: ["deleted"]` must also fail on "Deleted ORD-1008.".
|
|
34
|
+
|
|
35
|
+
## Built-in judges
|
|
36
|
+
|
|
37
|
+
Built into the CLI; `judges: {}` in `eval_config.yaml` (the scaffold default) uses them as they
|
|
38
|
+
are, and a `judges.<name>` entry overrides `scale`, `rubric` or `prompt_template` (versioned in the
|
|
39
|
+
repo, so a rubric change is a reviewed diff). A custom `prompt_template` may use exactly
|
|
40
|
+
`{metric}`, `{rubric}`, `{scale}`, `{conversation}`, `{transcript}`, `{response}`,
|
|
41
|
+
`{reference}`, `{context}`, `{reference_section}`, `{context_section}`, `{tool_calls_section}`;
|
|
42
|
+
any other placeholder is a configuration error (exit 3, when the config loads). The default
|
|
43
|
+
template asks for a JSON verdict with a `score`, which is why the template's `fake` judge answers
|
|
44
|
+
`{"score": 5, ...}`.
|
|
45
|
+
|
|
46
|
+
What every judge prompt contains, so a follow-up is judged in context:
|
|
47
|
+
|
|
48
|
+
- `{conversation}`: each earlier turn in full (the user message, every tool call the agent made
|
|
49
|
+
with its result, the agent's reply), then the latest user message. Dataset `system`/`assistant`
|
|
50
|
+
messages appear in place, marked as not sent to the agent.
|
|
51
|
+
- `{response}` and `{tool_calls_section}`: the reply being scored and its own tool calls.
|
|
52
|
+
- `{transcript}`: all of the above as one block, for custom templates.
|
|
53
|
+
- A tool result longer than `judge.max_tool_result_chars` (default 50000 characters; `null`
|
|
54
|
+
never cuts) ends in `[TRUNCATED by graph-agents-cli: the judge sees the first N of M characters
|
|
55
|
+
...]`, and the groundedness rubric tells the judge not to treat a claim from the omitted part as
|
|
56
|
+
unsupported. `eval grade` warns which cases were cut; raise the limit rather than the threshold.
|
|
57
|
+
|
|
58
|
+
| Metric | Scores | Needs |
|
|
59
|
+
|---|---|---|
|
|
60
|
+
| `response_quality` | overall helpfulness, correctness, tone against the rubric | nothing extra |
|
|
61
|
+
| `task_success` | whether the response accomplishes the case's task, counting what earlier turns already did; compares against `reference` when present | `reference` recommended |
|
|
62
|
+
| `groundedness` | whether every claim is supported by `context` or by the tool results of any turn | `context` or tool results |
|
|
63
|
+
|
|
64
|
+
Add a judge to a case with `"judge": {"task_success": {"threshold": 4}}`. Scores are integers on
|
|
65
|
+
the configured scale; `>= threshold` passes.
|
|
66
|
+
|
|
67
|
+
## Mandatory versus quality metrics
|
|
68
|
+
|
|
69
|
+
- A judge metric a case declares is **mandatory**: a score below threshold marks the case
|
|
70
|
+
`failed`.
|
|
71
|
+
- A judge metric listed under `quality_metrics:` is a **quality metric**: a score below its
|
|
72
|
+
`threshold` marks the case `quality_below_threshold` (not `failed`), and the run passes if, of
|
|
73
|
+
the cases scored on that metric, the fraction at or above the threshold is at least
|
|
74
|
+
`min_pass_rate`. Cases that do not declare the metric are not in the denominator (the results
|
|
75
|
+
record `scored` and `passed` per metric, and the table prints them); a quality metric no case
|
|
76
|
+
ran is `n/a (no case ran it)`.
|
|
77
|
+
- A judge call that raises marks the case `error` (exit 2); an unreachable judge, an unknown
|
|
78
|
+
metric or a metric with no threshold is a configuration error (exit 3). Judges are not called
|
|
79
|
+
for cases that already `failed` (deterministic checks come first).
|
|
80
|
+
- Deterministic checks can never be quality metrics.
|
|
81
|
+
- Judges run inside the project's environment: `eval grade` stages
|
|
82
|
+
`.graph-agents-cli/judge_runner.py` into the project and runs it with `uv run python`; the
|
|
83
|
+
runner calls `app.app_utils.model.get_judge_model()` with no arguments. Provider `fake` scores
|
|
84
|
+
every metric at the scale maximum, and `eval grade` then warns that the result is not a quality
|
|
85
|
+
signal.
|
|
86
|
+
|
|
87
|
+
```yaml
|
|
88
|
+
quality_metrics:
|
|
89
|
+
# at least 90 % of the cases scored on response_quality must score >= 4
|
|
90
|
+
response_quality: { threshold: 4, min_pass_rate: 0.9 }
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
## Custom metrics
|
|
94
|
+
|
|
95
|
+
```yaml
|
|
96
|
+
custom_metrics:
|
|
97
|
+
- tests.eval.metrics:answer_length_ok
|
|
98
|
+
```
|
|
99
|
+
|
|
100
|
+
```python
|
|
101
|
+
# tests/eval/metrics.py
|
|
102
|
+
def answer_length_ok(case: dict, trace: dict) -> dict:
|
|
103
|
+
words = len((trace.get("response") or "").split())
|
|
104
|
+
return {
|
|
105
|
+
"score": 1 if words <= case.get("metadata", {}).get("max_words", 200) else 0,
|
|
106
|
+
"reasoning": f"{words} words",
|
|
107
|
+
}
|
|
108
|
+
```
|
|
109
|
+
|
|
110
|
+
Signature `fn(case: dict, trace: dict) -> bool | number | {"score": n, "reasoning": str}`;
|
|
111
|
+
default threshold `1.0`. Custom metrics are config-level: they run for **every** case through
|
|
112
|
+
the judge runner inside the project's environment (so they may import project modules and call
|
|
113
|
+
`from app.app_utils.model import get_judge_model` for bespoke rubrics); a case's
|
|
114
|
+
`judge.<name>.threshold` only overrides the threshold. They follow the same mandatory/quality
|
|
115
|
+
rule as judge metrics (list one under `quality_metrics:` to allow a pass rate below 100 %).
|
|
116
|
+
|
|
117
|
+
## Judge configuration
|
|
118
|
+
|
|
119
|
+
Resolution order: `eval grade --judge-provider` / `--judge-model` -> `eval_config.yaml`
|
|
120
|
+
`judge: {provider, model}` -> `JUDGE_MODEL_PROVIDER` / `JUDGE_MODEL_NAME` / `JUDGE_BASE_URL` /
|
|
121
|
+
`JUDGE_API_KEY` -> the agent's `MODEL_*`. The CLI passes its choice to the runner as
|
|
122
|
+
`JUDGE_MODEL_PROVIDER` / `JUDGE_MODEL_NAME`, and the template's `get_judge_model()` reads them at
|
|
123
|
+
call time and builds the judge with `init_chat_model` like the agent. In the disconnected profile it is an on-network
|
|
124
|
+
OpenAI-compatible server. Prefer a judge at least as capable as the agent model; the same model
|
|
125
|
+
judging itself is acceptable for structural rubrics, weaker for quality. The judge reads every
|
|
126
|
+
turn of a case, so its context window must fit the conversation: `judge.max_tool_result_chars`
|
|
127
|
+
bounds each tool result, not the whole prompt.
|
|
128
|
+
|
|
129
|
+
## What to fix when a metric fails
|
|
130
|
+
|
|
131
|
+
| Symptom | Likely cause | Fix |
|
|
132
|
+
|---|---|---|
|
|
133
|
+
| `contains` fails but the answer is right | wording variance | use `regex` or a judge; or make the prompt ask for the phrase explicitly |
|
|
134
|
+
| `not_contains` fails on a correct refusal | the forbidden word also appears in a legitimate answer (a status list, a quoted request) | forbid the specific leak (an id, an address), not a common word |
|
|
135
|
+
| a multi-turn `tool_calls` check misses an earlier call | `scope` is `final_turn` (the default) | set `expect.scope: all_turns` |
|
|
136
|
+
| `tool_calls` fails: wrong tool | tool descriptions overlap | sharpen docstrings; remove unused tools |
|
|
137
|
+
| `tool_calls` fails: no call | model does not call tools | on `openai-compatible`, check the model supports tools and the server parses them |
|
|
138
|
+
| `tool_calls` fails: refused by policy | the call is outside `api-policy.yaml`, or over one of its `limits` | the operation must be allowed by the policy's owner (`graph-agents-cli api allow`, a reviewed change) and declared in `API_CALLS`, a limit raised deliberately (`api limits`), or the tool must change |
|
|
139
|
+
| `groundedness` low | the answer adds unsupported claims | instruct the model to cite tool results; return structured tool output |
|
|
140
|
+
| `groundedness` low and the case has `judge_notes` about a cut | a tool result was longer than `judge.max_tool_result_chars` | raise the limit (or `null`), or return a smaller result |
|
|
141
|
+
| `task_success` low with a `reference` | the reference is too specific | rewrite the reference as the essential content, not exact wording |
|
|
142
|
+
| `max_latency_ms` fails | tool or model slow | measure with `run -v`; cache; reduce context |
|
|
143
|
+
| `error` status | server crash, timeout, judge exception | read `reasons`; fix generation before grading |
|