vmware-debug 1.8.3__tar.gz → 1.8.5__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {vmware_debug-1.8.3 → vmware_debug-1.8.5}/PKG-INFO +2 -2
- {vmware_debug-1.8.3 → vmware_debug-1.8.5}/RELEASE_NOTES.md +157 -0
- {vmware_debug-1.8.3 → vmware_debug-1.8.5}/pyproject.toml +8 -2
- {vmware_debug-1.8.3 → vmware_debug-1.8.5}/server.json +2 -2
- vmware_debug-1.8.5/tests/eval/capability/__init__.py +0 -0
- vmware_debug-1.8.5/tests/eval/capability/_family.py +204 -0
- vmware_debug-1.8.5/tests/eval/capability/_scores.json +184 -0
- vmware_debug-1.8.5/tests/eval/capability/_scoring.py +245 -0
- vmware_debug-1.8.5/tests/eval/capability/_skill.py +62 -0
- vmware_debug-1.8.5/tests/eval/capability/conftest.py +80 -0
- vmware_debug-1.8.5/tests/eval/capability/test_entity_reachability.py +400 -0
- vmware_debug-1.8.5/tests/eval/capability/test_error_actionability.py +977 -0
- vmware_debug-1.8.5/tests/eval/capability/test_tool_description_quality.py +194 -0
- vmware_debug-1.8.5/tests/eval/capability/test_tool_manifest_budget.py +170 -0
- vmware_debug-1.8.5/tests/eval/regression/__init__.py +0 -0
- vmware_debug-1.8.5/tests/eval/regression/test_capability_grader.py +401 -0
- vmware_debug-1.8.5/tests/test_safe_error_passthrough.py +193 -0
- {vmware_debug-1.8.3 → vmware_debug-1.8.5}/uv.lock +71 -6
- {vmware_debug-1.8.3 → vmware_debug-1.8.5}/vmware_debug/__init__.py +1 -1
- {vmware_debug-1.8.3 → vmware_debug-1.8.5}/vmware_debug/envelope.py +94 -7
- {vmware_debug-1.8.3 → vmware_debug-1.8.5}/vmware_debug/mcp_server/server.py +85 -26
- {vmware_debug-1.8.3 → vmware_debug-1.8.5}/vmware_debug/ops/timeline.py +6 -2
- {vmware_debug-1.8.3 → vmware_debug-1.8.5}/.gitignore +0 -0
- {vmware_debug-1.8.3 → vmware_debug-1.8.5}/README-CN.md +0 -0
- {vmware_debug-1.8.3 → vmware_debug-1.8.5}/README.md +0 -0
- {vmware_debug-1.8.3 → vmware_debug-1.8.5}/SECURITY.md +0 -0
- {vmware_debug-1.8.3 → vmware_debug-1.8.5}/skills/vmware-debug/SKILL.md +0 -0
- {vmware_debug-1.8.3 → vmware_debug-1.8.5}/skills/vmware-debug/references/agent-guardrails.md +0 -0
- {vmware_debug-1.8.3 → vmware_debug-1.8.5}/skills/vmware-debug/references/capabilities.md +0 -0
- {vmware_debug-1.8.3 → vmware_debug-1.8.5}/skills/vmware-debug/references/cli-reference.md +0 -0
- {vmware_debug-1.8.3 → vmware_debug-1.8.5}/skills/vmware-debug/references/event-envelope.md +0 -0
- {vmware_debug-1.8.3 → vmware_debug-1.8.5}/skills/vmware-debug/references/routing.md +0 -0
- {vmware_debug-1.8.3 → vmware_debug-1.8.5}/skills/vmware-debug/references/setup-guide.md +0 -0
- {vmware_debug-1.8.3/tests/eval/regression → vmware_debug-1.8.5/tests/eval}/__init__.py +0 -0
- {vmware_debug-1.8.3 → vmware_debug-1.8.5}/tests/eval/regression/test_debug_regressions.py +0 -0
- {vmware_debug-1.8.3 → vmware_debug-1.8.5}/tests/eval/regression/test_declared_environment.py +0 -0
- {vmware_debug-1.8.3 → vmware_debug-1.8.5}/tests/eval/regression/test_read_only_mode.py +0 -0
- {vmware_debug-1.8.3 → vmware_debug-1.8.5}/tests/eval/regression/test_result_envelope.py +0 -0
- {vmware_debug-1.8.3 → vmware_debug-1.8.5}/tests/test_timeline.py +0 -0
- {vmware_debug-1.8.3 → vmware_debug-1.8.5}/vmware_debug/cli.py +0 -0
- {vmware_debug-1.8.3 → vmware_debug-1.8.5}/vmware_debug/mcp/__init__.py +0 -0
- {vmware_debug-1.8.3 → vmware_debug-1.8.5}/vmware_debug/mcp/tools.py +0 -0
- {vmware_debug-1.8.3 → vmware_debug-1.8.5}/vmware_debug/mcp_server/__init__.py +0 -0
- {vmware_debug-1.8.3 → vmware_debug-1.8.5}/vmware_debug/ops/__init__.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: vmware-debug
|
|
3
|
-
Version: 1.8.
|
|
3
|
+
Version: 1.8.5
|
|
4
4
|
Summary: VMware diagnostic brain — read-only incident triage, log/event correlation, and root-cause routing across the VMware skill family
|
|
5
5
|
Author-email: Wei Zhou <wei-wz.zhou@broadcom.com>
|
|
6
6
|
License-Expression: MIT
|
|
@@ -13,7 +13,7 @@ Requires-Python: >=3.10
|
|
|
13
13
|
Requires-Dist: mcp[cli]<2.0,>=1.10
|
|
14
14
|
Requires-Dist: rich<15.0,>=13.0
|
|
15
15
|
Requires-Dist: typer<1.0,>=0.12
|
|
16
|
-
Requires-Dist: vmware-policy<2.0,>=1.8.
|
|
16
|
+
Requires-Dist: vmware-policy<2.0,>=1.8.5
|
|
17
17
|
Description-Content-Type: text/markdown
|
|
18
18
|
|
|
19
19
|
<!-- mcp-name: io.github.zw008/vmware-debug -->
|
|
@@ -1,3 +1,160 @@
|
|
|
1
|
+
## v1.8.5 (2026-07-20) — the two fixes v1.8.4 announced now actually work
|
|
2
|
+
|
|
3
|
+
Four adversarial reviews of v1.8.4 found that both of its headline fixes were
|
|
4
|
+
incomplete in ways the release notes did not reflect. This release makes them
|
|
5
|
+
real. If you are on 1.8.4, this is the one to take.
|
|
6
|
+
|
|
7
|
+
### Fixed — a failure that was *returned* was still audited as a success
|
|
8
|
+
|
|
9
|
+
vmware-policy 1.8.4 added `report_tool_failure()` for tools that catch an
|
|
10
|
+
exception and return an error payload instead of raising. **No skill called it.**
|
|
11
|
+
|
|
12
|
+
Every string-returning tool therefore kept doing exactly what 1.8.4 said it had
|
|
13
|
+
stopped doing: writing `status=ok` to `~/.vmware/audit.db` for an operation that
|
|
14
|
+
failed, recording an undo token for a change that never happened, and telling the
|
|
15
|
+
circuit breaker the call succeeded so repeated failures never tripped it.
|
|
16
|
+
|
|
17
|
+
The surface this covered is not marginal:
|
|
18
|
+
|
|
19
|
+
| Skill | What was mis-audited |
|
|
20
|
+
|---|---|
|
|
21
|
+
| vmware-aiops | 25 of 49 tools, including **every undo-bearing write** — a failed `vm_power_on` left an undo token saying "power it back off" |
|
|
22
|
+
| vmware-avi | all 28 tools, including `vs_toggle` and `ako_restart` |
|
|
23
|
+
| vmware-storage | all 4 write tools |
|
|
24
|
+
| vmware-nsx | the 5 delete tools |
|
|
25
|
+
|
|
26
|
+
vmware-avi is worth calling out: before 1.8.4 its exceptions propagated and the
|
|
27
|
+
audit was correct. 1.8.4 caught them and returned a string, so **that release made
|
|
28
|
+
its audit trail worse than it had been.**
|
|
29
|
+
|
|
30
|
+
Skills whose tools already return dict payloads (vmware-monitor, vmware-vks,
|
|
31
|
+
vmware-aria, vmware-log-insight, vmware-harden, vmware-debug, vmware-pilot) were
|
|
32
|
+
already detected correctly. They gained a test proving it rather than a redundant
|
|
33
|
+
call.
|
|
34
|
+
|
|
35
|
+
### Fixed — narrowing `OSError` did not close the leak it was meant to close
|
|
36
|
+
|
|
37
|
+
1.8.4 narrowed the `_safe_error` passthrough because bare `OSError` let TLS and
|
|
38
|
+
DNS failures reach the agent with hostnames and certificate subjects in them.
|
|
39
|
+
That narrowing had no effect on the error it was written for:
|
|
40
|
+
|
|
41
|
+
```
|
|
42
|
+
ssl.SSLCertVerificationError → ssl.SSLError → OSError, ValueError
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
`ValueError` has been on every allowlist since long before 1.8.4, so a
|
|
46
|
+
certificate failure kept passing through — the commonest self-signed-certificate
|
|
47
|
+
failure in this family, carrying the hostname it was checked against. An
|
|
48
|
+
allowlist structurally cannot express "not this one".
|
|
49
|
+
|
|
50
|
+
Where `ssl.SSLError` can actually surface — the pyVmomi skills — it is now
|
|
51
|
+
reduced *ahead* of the allowlist. In the httpx skills TLS arrives wrapped as
|
|
52
|
+
`httpx.ConnectError`, and in vmware-avi as `requests.exceptions.SSLError`, so the
|
|
53
|
+
guard cannot fire there; in those skills the leak was the raw exception
|
|
54
|
+
interpolated into an already-allowlisted `*ApiError`, and that is now authored
|
|
55
|
+
text naming the config target and `verify_ssl` instead of the exception.
|
|
56
|
+
|
|
57
|
+
The missing-password error — this family's most common first-run failure, whose
|
|
58
|
+
entire remedy is the environment variable name it carries — keeps its message
|
|
59
|
+
through a narrow `ConfigError(OSError)` rather than the base class. Connection
|
|
60
|
+
failures are translated at the connection layer into an authored remedy that
|
|
61
|
+
names the target and the setting to change, with the raw detail left on
|
|
62
|
+
`__cause__` for the server log.
|
|
63
|
+
|
|
64
|
+
### Also fixed
|
|
65
|
+
|
|
66
|
+
- **vmware-vks**: the quickstart documented a password variable the code never
|
|
67
|
+
reads — following `README.md` verbatim produced "Password not found". Five
|
|
68
|
+
places, plus six references to a `doctor` command this CLI has never had, two
|
|
69
|
+
descriptions promising fields the tools do not return, and eight teaching
|
|
70
|
+
messages that `RuntimeError` was masking.
|
|
71
|
+
- **vmware-nsx**: an error cited `--route-advertisement`; the flag is `--advertise`.
|
|
72
|
+
- **vmware-pilot**: `get_workflow_status` told the model to call `approve` — a
|
|
73
|
+
tool the read-only gate withholds — as the required next step; and a hint
|
|
74
|
+
pointed at a filename that could never appear in that message.
|
|
75
|
+
- **vmware-aiops**: `vm_task_status` polling a *failed task* returned
|
|
76
|
+
`{"state": "error", "error": ...}` from a successful read, which the new
|
|
77
|
+
detection read as the call itself failing. The field is now `task_error`.
|
|
78
|
+
**This is a breaking change for anything parsing that payload.**
|
|
79
|
+
- Several remedies that were still being cut by the 300-character cap the 1.8.4
|
|
80
|
+
notes claimed to have addressed.
|
|
81
|
+
|
|
82
|
+
### Known and not fixed
|
|
83
|
+
|
|
84
|
+
`ConnectionError` remains one type from two sources in several skills — a
|
|
85
|
+
skill's own authored message and urllib3's `HTTPSConnectionPool(host=..., port=...)`
|
|
86
|
+
share it, and an allowlist cannot separate them. vmware-vks is converted; the
|
|
87
|
+
rest need their own domain type and are deferred rather than half-done.
|
|
88
|
+
|
|
89
|
+
## v1.8.4 (2026-07-20) — errors that teach, and tool descriptions a small model can route from
|
|
90
|
+
|
|
91
|
+
A capability eval was rolled out across the family and asked two open questions:
|
|
92
|
+
when a call fails, is the model told enough to fix it, and can it pick the right
|
|
93
|
+
tool from the description alone? Both answers were worse than anyone thought, and
|
|
94
|
+
in several places the reason was that the measurement was looking somewhere other
|
|
95
|
+
than where the model reads.
|
|
96
|
+
|
|
97
|
+
### Fixed — teaching messages were being discarded on the way to the agent
|
|
98
|
+
|
|
99
|
+
`_safe_error` reduces unrecognised exceptions to `"<Class>: operation failed."`
|
|
100
|
+
so raw API text, credentials in URLs and internal paths cannot reach an agent.
|
|
101
|
+
Its allowlist held only the builtin validation errors — so this skill's **own**
|
|
102
|
+
domain exceptions, the ones that exist precisely to carry a corrected next step,
|
|
103
|
+
had their messages replaced by their class names.
|
|
104
|
+
|
|
105
|
+
The effect was invisible from the CLI, which prints those messages in full.
|
|
106
|
+
|
|
107
|
+
The worst case was shared by nine skills: `config.py` raises exactly one
|
|
108
|
+
`OSError`, the missing-password error, whose entire remedy is the environment
|
|
109
|
+
variable name it names. An agent hitting an unconfigured target received
|
|
110
|
+
`OSError: operation failed.` and had nothing to act on. That is the family's most
|
|
111
|
+
common first-run failure, and it landed one release after the documented variable
|
|
112
|
+
names were corrected — so the message that would have unstuck the operator was
|
|
113
|
+
the one being thrown away.
|
|
114
|
+
|
|
115
|
+
The rule is now the property it always meant: **every exception this skill raises
|
|
116
|
+
on purpose passes through**, and only genuinely unplanned ones are reduced.
|
|
117
|
+
`RuntimeError` stays reduced — it is the generic catch-all and in several skills
|
|
118
|
+
carries raw upstream text.
|
|
119
|
+
|
|
120
|
+
### Fixed — error messages now carry the correction
|
|
121
|
+
|
|
122
|
+
Every message that reported a failure without saying how to recover was
|
|
123
|
+
rewritten: it names the offending value, gives an imperative remedy, and names
|
|
124
|
+
something concrete to act on — a tool that exists, a real CLI command, a config
|
|
125
|
+
file, an environment variable. Recovery becomes an instruction-following problem
|
|
126
|
+
rather than an inference one, which is what a weak model can still do.
|
|
127
|
+
|
|
128
|
+
Three classes of defect surfaced while doing it:
|
|
129
|
+
|
|
130
|
+
- **Remedies that were never delivered.** `_safe_error` truncates with no
|
|
131
|
+
ellipsis, so a message longer than the cap loses its closing sentence
|
|
132
|
+
silently. One message had been shipping at 396 characters against a 300-char
|
|
133
|
+
cap — its remedy had never once reached an agent. Messages now lead with the
|
|
134
|
+
remedy so a long interpolated value truncates the expendable detail instead.
|
|
135
|
+
- **Commands that do not exist.** One skill's error hints named a `doctor`
|
|
136
|
+
subcommand it does not have.
|
|
137
|
+
- **Tools that do not exist.** A tool description pointed at two sibling-skill
|
|
138
|
+
tools that had been renamed, and another named a tool that had moved to a
|
|
139
|
+
different skill entirely.
|
|
140
|
+
|
|
141
|
+
### Improved — tool descriptions state when to use them and what to call next
|
|
142
|
+
|
|
143
|
+
The description is the API for a small model: an unstated routing rule is a
|
|
144
|
+
routing rule that does not exist, and a tool with no stated next hop is one the
|
|
145
|
+
model stops at. Descriptions now say when to prefer this tool over a sibling,
|
|
146
|
+
what shape comes back, the caveat that bites, and which tool to call after.
|
|
147
|
+
|
|
148
|
+
**Manifest size did not grow.** Descriptions load into every session, so the
|
|
149
|
+
routing clauses were paid for by cutting duplicated reference material —
|
|
150
|
+
repeated boilerplate, examples that restated the parameter list, and prose
|
|
151
|
+
copies of the pagination contract.
|
|
152
|
+
|
|
153
|
+
### Note
|
|
154
|
+
|
|
155
|
+
Every tool and CLI command named anywhere in this release was verified against
|
|
156
|
+
the live MCP registry and the live command tree, not against documentation.
|
|
157
|
+
|
|
1
158
|
## v1.8.3 (2026-07-20) — credentials resolve as a pair; documented env vars now exist
|
|
2
159
|
|
|
3
160
|
### Changed — version alignment
|
|
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "vmware-debug"
|
|
7
|
-
version = "1.8.
|
|
7
|
+
version = "1.8.5"
|
|
8
8
|
description = "VMware diagnostic brain — read-only incident triage, log/event correlation, and root-cause routing across the VMware skill family"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = "MIT"
|
|
@@ -21,7 +21,7 @@ dependencies = [
|
|
|
21
21
|
"typer>=0.12,<1.0",
|
|
22
22
|
"rich>=13.0,<15.0",
|
|
23
23
|
"mcp[cli]>=1.10,<2.0",
|
|
24
|
-
"vmware-policy>=1.8.
|
|
24
|
+
"vmware-policy>=1.8.5,<2.0",
|
|
25
25
|
]
|
|
26
26
|
|
|
27
27
|
[project.scripts]
|
|
@@ -31,6 +31,12 @@ vmware-debug-mcp = "vmware_debug.mcp_server.server:main"
|
|
|
31
31
|
[tool.hatch.build.targets.wheel]
|
|
32
32
|
packages = ["vmware_debug"]
|
|
33
33
|
|
|
34
|
+
[tool.pytest.ini_options]
|
|
35
|
+
testpaths = ["tests"]
|
|
36
|
+
markers = [
|
|
37
|
+
"capability: Capability evals — scored trends, not pass/fail gates. Excluded from the default run (they measure quality, so <100% is expected and must not read as a broken build). Run them with: pytest -m capability",
|
|
38
|
+
]
|
|
39
|
+
addopts = "-m 'not capability'"
|
|
34
40
|
[dependency-groups]
|
|
35
41
|
dev = [
|
|
36
42
|
"pytest>=8.0,<10.0",
|
|
@@ -7,12 +7,12 @@
|
|
|
7
7
|
"url": "https://github.com/zw008/VMware-Debug",
|
|
8
8
|
"source": "github"
|
|
9
9
|
},
|
|
10
|
-
"version": "1.8.
|
|
10
|
+
"version": "1.8.5",
|
|
11
11
|
"packages": [
|
|
12
12
|
{
|
|
13
13
|
"registryType": "pypi",
|
|
14
14
|
"identifier": "vmware-debug",
|
|
15
|
-
"version": "1.8.
|
|
15
|
+
"version": "1.8.5",
|
|
16
16
|
"transport": {
|
|
17
17
|
"type": "stdio"
|
|
18
18
|
}
|
|
File without changes
|
|
@@ -0,0 +1,204 @@
|
|
|
1
|
+
"""What every skill in the family exposes — generated, do not edit.
|
|
2
|
+
|
|
3
|
+
Written by scripts/install_capability_evals.sh, which is the only place
|
|
4
|
+
that can see all twelve repos at once. Error messages and tool
|
|
5
|
+
descriptions route across skills, so checking those citations needs the
|
|
6
|
+
sibling surfaces; deriving them here beats a hand-kept list that goes
|
|
7
|
+
stale without saying so.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
FAMILY_TOOLS: dict[str, frozenset[str]] = {
|
|
13
|
+
'vmware-aiops': frozenset({
|
|
14
|
+
'acknowledge_vcenter_alarm', 'attach_iso_to_vm', 'batch_clone_vms',
|
|
15
|
+
'batch_deploy_from_spec', 'batch_linked_clone_vms', 'browse_datastore',
|
|
16
|
+
'cluster_add_host', 'cluster_configure', 'cluster_create', 'cluster_delete',
|
|
17
|
+
'cluster_health_summary', 'cluster_info', 'cluster_remove_host',
|
|
18
|
+
'convert_vm_to_template', 'cross_vcenter_attention',
|
|
19
|
+
'datastore_investigation_bundle', 'deploy_linked_clone', 'deploy_vm_from_ova',
|
|
20
|
+
'deploy_vm_from_template', 'host_investigation_bundle', 'list_vcenter_alarms',
|
|
21
|
+
'reset_vcenter_alarm', 'scan_datastore_images', 'vm_apply_plan', 'vm_cancel_ttl',
|
|
22
|
+
'vm_clean_slate', 'vm_clone', 'vm_create', 'vm_create_plan', 'vm_create_snapshot',
|
|
23
|
+
'vm_delete', 'vm_delete_snapshot', 'vm_guest_download', 'vm_guest_exec',
|
|
24
|
+
'vm_guest_exec_output', 'vm_guest_provision', 'vm_guest_upload',
|
|
25
|
+
'vm_investigation_bundle', 'vm_list_plans', 'vm_list_snapshots', 'vm_list_ttl',
|
|
26
|
+
'vm_migrate', 'vm_power_off', 'vm_power_on', 'vm_reconfigure', 'vm_revert_snapshot',
|
|
27
|
+
'vm_rollback_plan', 'vm_set_ttl', 'vm_task_status',
|
|
28
|
+
}),
|
|
29
|
+
'vmware-aria': frozenset({
|
|
30
|
+
'acknowledge_alert', 'cancel_alert', 'create_alert_definition',
|
|
31
|
+
'delete_alert_definition', 'delete_report', 'generate_report', 'get_alert',
|
|
32
|
+
'get_aria_health', 'get_capacity_overview', 'get_remaining_capacity', 'get_report',
|
|
33
|
+
'get_resource', 'get_resource_health', 'get_resource_metrics',
|
|
34
|
+
'get_resource_riskbadge', 'get_time_remaining', 'get_top_consumers',
|
|
35
|
+
'investigate_alert', 'list_alert_definitions', 'list_alerts', 'list_anomalies',
|
|
36
|
+
'list_collector_groups', 'list_report_definitions', 'list_reports',
|
|
37
|
+
'list_resources', 'list_rightsizing_recommendations', 'list_symptom_definitions',
|
|
38
|
+
'set_alert_definition_state',
|
|
39
|
+
}),
|
|
40
|
+
'vmware-avi': frozenset({
|
|
41
|
+
'ako_amko_status', 'ako_clusters', 'ako_config_diff', 'ako_config_show',
|
|
42
|
+
'ako_config_upgrade', 'ako_ingress_check', 'ako_ingress_diagnose',
|
|
43
|
+
'ako_ingress_map', 'ako_logs', 'ako_restart', 'ako_status', 'ako_sync_diff',
|
|
44
|
+
'ako_sync_force', 'ako_sync_status', 'ako_version', 'pool_list',
|
|
45
|
+
'pool_member_disable', 'pool_member_enable', 'pool_members', 'se_health', 'se_list',
|
|
46
|
+
'ssl_expiry_check', 'ssl_list', 'vs_analytics', 'vs_error_logs', 'vs_list',
|
|
47
|
+
'vs_status', 'vs_toggle',
|
|
48
|
+
}),
|
|
49
|
+
'vmware-debug': frozenset({
|
|
50
|
+
'incident_timeline', 'list_symptom_categories',
|
|
51
|
+
}),
|
|
52
|
+
'vmware-harden': frozenset({
|
|
53
|
+
'get_baseline_rules', 'get_remediation', 'list_baselines', 'list_drift_events',
|
|
54
|
+
'list_violations', 'scan_target',
|
|
55
|
+
}),
|
|
56
|
+
'vmware-log-insight': frozenset({
|
|
57
|
+
'alert_get', 'alert_history', 'alert_list', 'log_aggregate', 'log_fields',
|
|
58
|
+
'log_search', 'log_version',
|
|
59
|
+
}),
|
|
60
|
+
'vmware-monitor': frozenset({
|
|
61
|
+
'active_sessions', 'active_tasks', 'certificate_status', 'cluster_health_summary',
|
|
62
|
+
'cross_vcenter_attention', 'datastore_capacity', 'datastore_investigation_bundle',
|
|
63
|
+
'get_alarms', 'get_events', 'get_host_sensors', 'get_host_services',
|
|
64
|
+
'host_investigation_bundle', 'host_log_scan', 'host_performance', 'license_status',
|
|
65
|
+
'list_all_clusters', 'list_all_datastores', 'list_all_networks', 'list_esxi_hosts',
|
|
66
|
+
'list_virtual_machines', 'ntp_status', 'resource_pool_usage', 'snapshot_aging',
|
|
67
|
+
'vm_info', 'vm_investigation_bundle', 'vm_list_snapshots', 'vm_performance',
|
|
68
|
+
}),
|
|
69
|
+
'vmware-nsx': frozenset({
|
|
70
|
+
'configure_tier0_bgp', 'create_ip_pool', 'create_nat_rule', 'create_segment',
|
|
71
|
+
'create_static_route', 'create_tier1_gateway', 'delete_ip_pool', 'delete_nat_rule',
|
|
72
|
+
'delete_segment', 'delete_static_route', 'delete_tier1_gateway',
|
|
73
|
+
'get_bgp_neighbors', 'get_edge_cluster_status', 'get_ip_pool_usage',
|
|
74
|
+
'get_logical_port_status', 'get_nsx_manager_status', 'get_segment',
|
|
75
|
+
'get_segment_port_for_vm', 'get_tier0_gateway', 'get_tier1_gateway',
|
|
76
|
+
'get_transport_node_status', 'list_edge_clusters', 'list_ip_pools',
|
|
77
|
+
'list_nat_rules', 'list_nsx_alarms', 'list_segments', 'list_static_routes',
|
|
78
|
+
'list_tier0_gateways', 'list_tier1_gateways', 'list_transport_nodes',
|
|
79
|
+
'list_transport_zones', 'update_segment', 'update_tier1_gateway',
|
|
80
|
+
}),
|
|
81
|
+
'vmware-nsx-security': frozenset({
|
|
82
|
+
'apply_vm_tag', 'create_dfw_policy', 'create_dfw_rule', 'create_group',
|
|
83
|
+
'delete_dfw_policy', 'delete_dfw_rule', 'delete_group', 'get_dfw_policy',
|
|
84
|
+
'get_dfw_rule_stats', 'get_group', 'get_idps_status', 'get_traceflow_result',
|
|
85
|
+
'list_dfw_policies', 'list_dfw_rules', 'list_groups', 'list_idps_profiles',
|
|
86
|
+
'list_vm_tags', 'remove_vm_tag', 'run_traceflow', 'update_dfw_policy',
|
|
87
|
+
'update_dfw_rule',
|
|
88
|
+
}),
|
|
89
|
+
'vmware-pilot': frozenset({
|
|
90
|
+
'approve', 'cancel_workflow', 'confirm_draft', 'create_workflow', 'design_workflow',
|
|
91
|
+
'get_skill_catalog', 'get_workflow_status', 'list_workflows', 'plan_workflow',
|
|
92
|
+
'review_workflow', 'rollback', 'run_workflow', 'update_draft',
|
|
93
|
+
}),
|
|
94
|
+
'vmware-storage': frozenset({
|
|
95
|
+
'browse_datastore', 'list_all_datastores', 'list_cached_images',
|
|
96
|
+
'scan_datastore_images', 'storage_iscsi_add_target', 'storage_iscsi_enable',
|
|
97
|
+
'storage_iscsi_remove_target', 'storage_iscsi_status', 'storage_rescan',
|
|
98
|
+
'vsan_capacity', 'vsan_health',
|
|
99
|
+
}),
|
|
100
|
+
'vmware-vks': frozenset({
|
|
101
|
+
'check_vks_compatibility', 'create_namespace', 'create_tkc_cluster',
|
|
102
|
+
'delete_namespace', 'delete_tkc_cluster', 'get_harbor_info', 'get_namespace',
|
|
103
|
+
'get_supervisor_kubeconfig', 'get_supervisor_status', 'get_tkc_available_versions',
|
|
104
|
+
'get_tkc_cluster', 'get_tkc_kubeconfig', 'list_namespace_storage_usage',
|
|
105
|
+
'list_namespaces', 'list_supervisor_storage_policies', 'list_tkc_clusters',
|
|
106
|
+
'list_vm_classes', 'scale_tkc_cluster', 'update_namespace', 'upgrade_tkc_cluster',
|
|
107
|
+
}),
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
FAMILY_COMMANDS: dict[str, frozenset[str]] = {
|
|
111
|
+
'vmware-aiops': frozenset({
|
|
112
|
+
'alarm acknowledge', 'alarm list', 'alarm reset', 'attention', 'cluster add-host',
|
|
113
|
+
'cluster configure', 'cluster create', 'cluster delete', 'cluster info',
|
|
114
|
+
'cluster remove-host', 'daemon start', 'daemon status', 'daemon stop',
|
|
115
|
+
'datastore browse', 'datastore scan-images', 'deploy batch', 'deploy batch-clone',
|
|
116
|
+
'deploy iso', 'deploy linked-clone', 'deploy mark-template', 'deploy ova',
|
|
117
|
+
'deploy template', 'doctor', 'hub status', 'init', 'investigate datastore',
|
|
118
|
+
'investigate host', 'investigate vm', 'mcp', 'mcp-config generate',
|
|
119
|
+
'mcp-config install', 'mcp-config list', 'plan list', 'scan now', 'summary',
|
|
120
|
+
'vm cancel-ttl', 'vm clean-slate', 'vm clone', 'vm create', 'vm delete',
|
|
121
|
+
'vm guest-download', 'vm guest-exec', 'vm guest-upload', 'vm list-ttl',
|
|
122
|
+
'vm migrate', 'vm power-off', 'vm power-on', 'vm reconfigure', 'vm set-ttl',
|
|
123
|
+
'vm snapshot-create', 'vm snapshot-delete', 'vm snapshot-list',
|
|
124
|
+
'vm snapshot-revert', 'vm task-status',
|
|
125
|
+
}),
|
|
126
|
+
'vmware-aria': frozenset({
|
|
127
|
+
'alert acknowledge', 'alert cancel', 'alert definitions', 'alert get', 'alert list',
|
|
128
|
+
'anomaly list', 'anomaly risk', 'capacity overview', 'capacity remaining',
|
|
129
|
+
'capacity rightsizing', 'capacity time-remaining', 'doctor', 'health collectors',
|
|
130
|
+
'health status', 'init', 'mcp', 'report definitions', 'report delete',
|
|
131
|
+
'report generate', 'report get', 'report list', 'resource get', 'resource health',
|
|
132
|
+
'resource list', 'resource metrics', 'resource top',
|
|
133
|
+
}),
|
|
134
|
+
'vmware-avi': frozenset({
|
|
135
|
+
'ako amko-status', 'ako clusters', 'ako config-diff', 'ako config-show',
|
|
136
|
+
'ako config-upgrade', 'ako ingress-check', 'ako ingress-diagnose',
|
|
137
|
+
'ako ingress-map', 'ako logs', 'ako restart', 'ako status', 'ako sync-diff',
|
|
138
|
+
'ako sync-force', 'ako sync-status', 'ako version', 'analytics', 'config', 'doctor',
|
|
139
|
+
'init', 'logs', 'mcp', 'pool disable', 'pool enable', 'pool members', 'se health',
|
|
140
|
+
'se list', 'ssl expiry', 'ssl list', 'vs disable', 'vs enable', 'vs list',
|
|
141
|
+
'vs status',
|
|
142
|
+
}),
|
|
143
|
+
'vmware-debug': frozenset({
|
|
144
|
+
'categories', 'mcp', 'triage', 'version',
|
|
145
|
+
}),
|
|
146
|
+
'vmware-harden': frozenset({
|
|
147
|
+
'advise', 'apply', 'baseline import', 'baseline list', 'baseline validate',
|
|
148
|
+
'doctor', 'drift', 'mcp', 'report', 'scan', 'web',
|
|
149
|
+
}),
|
|
150
|
+
'vmware-log-insight': frozenset({
|
|
151
|
+
'aggregate', 'alert get', 'alert history', 'alert list', 'doctor', 'fields', 'mcp',
|
|
152
|
+
'search', 'version',
|
|
153
|
+
}),
|
|
154
|
+
'vmware-monitor': frozenset({
|
|
155
|
+
'activity sessions', 'activity tasks', 'attention', 'capacity datastores',
|
|
156
|
+
'capacity pools', 'daemon start', 'daemon status', 'daemon stop', 'doctor',
|
|
157
|
+
'health alarms', 'health events', 'health sensors', 'health services',
|
|
158
|
+
'infra certs', 'infra licenses', 'infra ntp', 'init', 'inventory clusters',
|
|
159
|
+
'inventory datastores', 'inventory hosts', 'inventory networks', 'inventory vms',
|
|
160
|
+
'investigate datastore', 'investigate host', 'investigate vm', 'mcp',
|
|
161
|
+
'mcp-config generate', 'mcp-config list', 'perf hosts', 'perf vms', 'scan now',
|
|
162
|
+
'snapshots aging', 'summary', 'vm info', 'vm snapshot-list',
|
|
163
|
+
}),
|
|
164
|
+
'vmware-nsx': frozenset({
|
|
165
|
+
'doctor', 'gateway configure-tier0-bgp', 'gateway create-tier1',
|
|
166
|
+
'gateway delete-tier1', 'gateway update-tier1', 'health alarms',
|
|
167
|
+
'health edge-cluster-status', 'health manager-status',
|
|
168
|
+
'health transport-node-status', 'init', 'inventory get-segment',
|
|
169
|
+
'inventory get-tier0', 'inventory get-tier1', 'inventory list-edge-clusters',
|
|
170
|
+
'inventory list-segments', 'inventory list-tier0s', 'inventory list-tier1s',
|
|
171
|
+
'inventory list-transport-nodes', 'inventory list-transport-zones',
|
|
172
|
+
'ip-pool create', 'ip-pool delete', 'mcp', 'mcp-config generate',
|
|
173
|
+
'mcp-config install', 'mcp-config list', 'nat create-rule', 'nat delete-rule',
|
|
174
|
+
'networking bgp-neighbors', 'networking ip-pool-usage', 'networking list-ip-pools',
|
|
175
|
+
'networking list-nat-rules', 'networking list-static-routes', 'route create-static',
|
|
176
|
+
'route delete-static', 'segment create', 'segment delete', 'segment update',
|
|
177
|
+
'troubleshoot port-status', 'troubleshoot vm-segment',
|
|
178
|
+
}),
|
|
179
|
+
'vmware-nsx-security': frozenset({
|
|
180
|
+
'doctor', 'group delete', 'group get', 'group list', 'idps profiles', 'idps status',
|
|
181
|
+
'init', 'mcp', 'policy create', 'policy delete', 'policy get', 'policy list',
|
|
182
|
+
'rule delete', 'rule list', 'rule stats', 'tag apply', 'tag list', 'tag remove',
|
|
183
|
+
'traceflow run',
|
|
184
|
+
}),
|
|
185
|
+
'vmware-pilot': frozenset({
|
|
186
|
+
'mcp', 'version',
|
|
187
|
+
}),
|
|
188
|
+
'vmware-storage': frozenset({
|
|
189
|
+
'datastore browse', 'datastore list', 'datastore scan-images', 'doctor', 'init',
|
|
190
|
+
'iscsi add-target', 'iscsi enable', 'iscsi remove-target', 'iscsi rescan',
|
|
191
|
+
'iscsi status', 'mcp', 'vsan capacity', 'vsan health',
|
|
192
|
+
}),
|
|
193
|
+
'vmware-vks': frozenset({
|
|
194
|
+
'check', 'harbor', 'init', 'kubeconfig get', 'kubeconfig supervisor', 'mcp',
|
|
195
|
+
'namespace create', 'namespace delete', 'namespace get', 'namespace list',
|
|
196
|
+
'namespace update', 'namespace vm-classes', 'preflight-auth', 'storage',
|
|
197
|
+
'supervisor status', 'supervisor storage-policies', 'tkc create', 'tkc delete',
|
|
198
|
+
'tkc get', 'tkc list', 'tkc scale', 'tkc upgrade', 'tkc versions',
|
|
199
|
+
}),
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
#: Every tool name anywhere in the family, for citation checks that do
|
|
203
|
+
#: not care which skill owns the name.
|
|
204
|
+
ALL_TOOLS: frozenset[str] = frozenset().union(*FAMILY_TOOLS.values())
|
|
@@ -0,0 +1,184 @@
|
|
|
1
|
+
{
|
|
2
|
+
"_comment": "Capability eval scores. Regenerate with: pytest -m capability. These are tracked trends, not pass/fail gates \u2014 see _scoring.py. Entries marked stale=true were carried over from an earlier run because this run did not measure them.",
|
|
3
|
+
"scores": {
|
|
4
|
+
"entity_reachability_full": {
|
|
5
|
+
"detail": {
|
|
6
|
+
"broken_chains": [],
|
|
7
|
+
"entry_points": [
|
|
8
|
+
"incident_timeline",
|
|
9
|
+
"list_symptom_categories"
|
|
10
|
+
],
|
|
11
|
+
"reachable_on_surface": 0,
|
|
12
|
+
"reachable_only_via_description": 0,
|
|
13
|
+
"required_entity_params": 0,
|
|
14
|
+
"required_params_classified": "0/1",
|
|
15
|
+
"tool_count": 2
|
|
16
|
+
},
|
|
17
|
+
"maximum": 1,
|
|
18
|
+
"pct": 100.0,
|
|
19
|
+
"unit": "required entity params",
|
|
20
|
+
"value": 1
|
|
21
|
+
},
|
|
22
|
+
"entity_reachability_read_only": {
|
|
23
|
+
"detail": {
|
|
24
|
+
"broken_chains": [],
|
|
25
|
+
"entry_points": [
|
|
26
|
+
"incident_timeline",
|
|
27
|
+
"list_symptom_categories"
|
|
28
|
+
],
|
|
29
|
+
"reachable_on_surface": 0,
|
|
30
|
+
"reachable_only_via_description": 0,
|
|
31
|
+
"required_entity_params": 0,
|
|
32
|
+
"required_params_classified": "0/1",
|
|
33
|
+
"tool_count": 2
|
|
34
|
+
},
|
|
35
|
+
"maximum": 1,
|
|
36
|
+
"pct": 100.0,
|
|
37
|
+
"unit": "required entity params",
|
|
38
|
+
"value": 1
|
|
39
|
+
},
|
|
40
|
+
"entry_point_availability": {
|
|
41
|
+
"detail": {
|
|
42
|
+
"full_entry_points": 2,
|
|
43
|
+
"read_only_entry_points": 2
|
|
44
|
+
},
|
|
45
|
+
"maximum": 2,
|
|
46
|
+
"pct": 100.0,
|
|
47
|
+
"unit": "entry points",
|
|
48
|
+
"value": 2
|
|
49
|
+
},
|
|
50
|
+
"error_actionability": {
|
|
51
|
+
"detail": {
|
|
52
|
+
"composed_hint_sites": [],
|
|
53
|
+
"dead_end_count": 0,
|
|
54
|
+
"dead_end_errors": [],
|
|
55
|
+
"per_dimension_pct": {
|
|
56
|
+
"names_artifact": 100.0,
|
|
57
|
+
"names_input": 100.0,
|
|
58
|
+
"states_remedy": 100.0
|
|
59
|
+
},
|
|
60
|
+
"raise_sites": 7
|
|
61
|
+
},
|
|
62
|
+
"maximum": 21,
|
|
63
|
+
"pct": 100.0,
|
|
64
|
+
"unit": "points",
|
|
65
|
+
"value": 21
|
|
66
|
+
},
|
|
67
|
+
"manifest_context_headroom": {
|
|
68
|
+
"detail": {
|
|
69
|
+
"heaviest_tools": {
|
|
70
|
+
"incident_timeline": 329,
|
|
71
|
+
"list_symptom_categories": 121
|
|
72
|
+
},
|
|
73
|
+
"manifest_tokens": 450,
|
|
74
|
+
"mean_tokens_per_tool": 225.0,
|
|
75
|
+
"reference_context": 16384,
|
|
76
|
+
"tool_count": 2
|
|
77
|
+
},
|
|
78
|
+
"maximum": 16384,
|
|
79
|
+
"pct": 97.3,
|
|
80
|
+
"unit": "tokens",
|
|
81
|
+
"value": 15934
|
|
82
|
+
},
|
|
83
|
+
"parameter_documentation_coverage": {
|
|
84
|
+
"detail": {
|
|
85
|
+
"undocumented_by_tool": {}
|
|
86
|
+
},
|
|
87
|
+
"maximum": 4,
|
|
88
|
+
"pct": 100.0,
|
|
89
|
+
"unit": "parameters",
|
|
90
|
+
"value": 4
|
|
91
|
+
},
|
|
92
|
+
"per_tool_token_discipline": {
|
|
93
|
+
"detail": {
|
|
94
|
+
"ceiling": 350,
|
|
95
|
+
"over_ceiling": {}
|
|
96
|
+
},
|
|
97
|
+
"maximum": 2,
|
|
98
|
+
"pct": 100.0,
|
|
99
|
+
"unit": "tools",
|
|
100
|
+
"value": 2
|
|
101
|
+
},
|
|
102
|
+
"read_only_manifest_saving": {
|
|
103
|
+
"detail": {
|
|
104
|
+
"full_manifest_tokens": 450,
|
|
105
|
+
"full_tool_count": 2,
|
|
106
|
+
"gated_manifest_tokens": 450,
|
|
107
|
+
"gated_tool_count": 2,
|
|
108
|
+
"withheld_tools": []
|
|
109
|
+
},
|
|
110
|
+
"maximum": 450,
|
|
111
|
+
"pct": 0.0,
|
|
112
|
+
"unit": "tokens",
|
|
113
|
+
"value": 0
|
|
114
|
+
},
|
|
115
|
+
"read_write_marker_coverage": {
|
|
116
|
+
"detail": {
|
|
117
|
+
"unmarked": []
|
|
118
|
+
},
|
|
119
|
+
"maximum": 2,
|
|
120
|
+
"pct": 100.0,
|
|
121
|
+
"unit": "points",
|
|
122
|
+
"value": 2
|
|
123
|
+
},
|
|
124
|
+
"remedy_survives_truncation": {
|
|
125
|
+
"detail": {
|
|
126
|
+
"budget_chars": 500,
|
|
127
|
+
"over_budget": [],
|
|
128
|
+
"remedy_after_interpolation": [
|
|
129
|
+
"envelope.py:135",
|
|
130
|
+
"envelope.py:95",
|
|
131
|
+
"envelope.py:160",
|
|
132
|
+
"envelope.py:107",
|
|
133
|
+
"envelope.py:200",
|
|
134
|
+
"timeline.py:119"
|
|
135
|
+
]
|
|
136
|
+
},
|
|
137
|
+
"maximum": 7,
|
|
138
|
+
"pct": 100.0,
|
|
139
|
+
"unit": "messages",
|
|
140
|
+
"value": 7
|
|
141
|
+
},
|
|
142
|
+
"teaching_error_rate": {
|
|
143
|
+
"detail": {},
|
|
144
|
+
"maximum": 7,
|
|
145
|
+
"pct": 100.0,
|
|
146
|
+
"unit": "messages",
|
|
147
|
+
"value": 7
|
|
148
|
+
},
|
|
149
|
+
"tool_description_quality": {
|
|
150
|
+
"detail": {
|
|
151
|
+
"per_dimension_pct": {
|
|
152
|
+
"args": 100.0,
|
|
153
|
+
"gotcha": 100.0,
|
|
154
|
+
"marker": 100.0,
|
|
155
|
+
"next_hop": 100.0,
|
|
156
|
+
"what": 100.0,
|
|
157
|
+
"when": 100.0
|
|
158
|
+
},
|
|
159
|
+
"tools_graded": 2,
|
|
160
|
+
"weakest_tools": {
|
|
161
|
+
"incident_timeline": [],
|
|
162
|
+
"list_symptom_categories": []
|
|
163
|
+
}
|
|
164
|
+
},
|
|
165
|
+
"maximum": 12,
|
|
166
|
+
"pct": 100.0,
|
|
167
|
+
"unit": "points",
|
|
168
|
+
"value": 12
|
|
169
|
+
},
|
|
170
|
+
"tool_failure_payload_quality": {
|
|
171
|
+
"detail": {
|
|
172
|
+
"bare_string_error_lines": [],
|
|
173
|
+
"carries_hint": 1,
|
|
174
|
+
"dict_shaped": 1,
|
|
175
|
+
"error_return_sites": 1,
|
|
176
|
+
"hint_names_artifact": 1
|
|
177
|
+
},
|
|
178
|
+
"maximum": 3,
|
|
179
|
+
"pct": 100.0,
|
|
180
|
+
"unit": "checks",
|
|
181
|
+
"value": 3
|
|
182
|
+
}
|
|
183
|
+
}
|
|
184
|
+
}
|