itsmbench-audit 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- itsmbench_audit-0.1.0/.gitignore +86 -0
- itsmbench_audit-0.1.0/LICENSE +21 -0
- itsmbench_audit-0.1.0/PKG-INFO +384 -0
- itsmbench_audit-0.1.0/README.md +346 -0
- itsmbench_audit-0.1.0/itsmbench_audit/__init__.py +1 -0
- itsmbench_audit-0.1.0/itsmbench_audit/__main__.py +2882 -0
- itsmbench_audit-0.1.0/pyproject.toml +40 -0
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
# Python
|
|
2
|
+
__pycache__/
|
|
3
|
+
*.py[cod]
|
|
4
|
+
*$py.class
|
|
5
|
+
*.pyo
|
|
6
|
+
*.pyd
|
|
7
|
+
*.egg
|
|
8
|
+
*.egg-info/
|
|
9
|
+
dist/
|
|
10
|
+
build/
|
|
11
|
+
eggs/
|
|
12
|
+
parts/
|
|
13
|
+
var/
|
|
14
|
+
sdist/
|
|
15
|
+
wheels/
|
|
16
|
+
*.egg-link
|
|
17
|
+
pip-wheel-metadata/
|
|
18
|
+
share/python-wheels/
|
|
19
|
+
MANIFEST
|
|
20
|
+
itsmbench_audit/__pycache__
|
|
21
|
+
|
|
22
|
+
# Virtual environments
|
|
23
|
+
.venv/
|
|
24
|
+
venv/
|
|
25
|
+
env/
|
|
26
|
+
ENV/
|
|
27
|
+
.env
|
|
28
|
+
|
|
29
|
+
# Distribution / packaging
|
|
30
|
+
.Python
|
|
31
|
+
develop-eggs/
|
|
32
|
+
lib/
|
|
33
|
+
lib64/
|
|
34
|
+
.installed.cfg
|
|
35
|
+
|
|
36
|
+
# Installer logs
|
|
37
|
+
pip-log.txt
|
|
38
|
+
pip-delete-this-directory.txt
|
|
39
|
+
|
|
40
|
+
# Unit test / coverage
|
|
41
|
+
htmlcov/
|
|
42
|
+
.tox/
|
|
43
|
+
.nox/
|
|
44
|
+
.coverage
|
|
45
|
+
.coverage.*
|
|
46
|
+
.cache
|
|
47
|
+
nosetests.xml
|
|
48
|
+
coverage.xml
|
|
49
|
+
*.cover
|
|
50
|
+
*.py,cover
|
|
51
|
+
.hypothesis/
|
|
52
|
+
.pytest_cache/
|
|
53
|
+
pytestdebug.log
|
|
54
|
+
|
|
55
|
+
# Type checking
|
|
56
|
+
.mypy_cache/
|
|
57
|
+
.dmypy.json
|
|
58
|
+
dmypy.json
|
|
59
|
+
.pytype/
|
|
60
|
+
.pyre/
|
|
61
|
+
|
|
62
|
+
# Ruff cache
|
|
63
|
+
.ruff_cache/
|
|
64
|
+
|
|
65
|
+
# IDEs
|
|
66
|
+
.vscode/
|
|
67
|
+
.idea/
|
|
68
|
+
*.swp
|
|
69
|
+
*.swo
|
|
70
|
+
*~
|
|
71
|
+
.DS_Store
|
|
72
|
+
Thumbs.db
|
|
73
|
+
|
|
74
|
+
# Jupyter
|
|
75
|
+
.ipynb_checkpoints/
|
|
76
|
+
*.ipynb
|
|
77
|
+
|
|
78
|
+
# Corpus / large data outputs
|
|
79
|
+
corpus_output.txt
|
|
80
|
+
*.jsonl
|
|
81
|
+
*.parquet
|
|
82
|
+
|
|
83
|
+
# Secrets / local config
|
|
84
|
+
.env.local
|
|
85
|
+
.env.*.local
|
|
86
|
+
secrets.toml
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 ITSMBench Contributors
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,384 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: itsmbench-audit
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Static preflight audit tooling for ITSMBench evaluations.
|
|
5
|
+
License: MIT License
|
|
6
|
+
|
|
7
|
+
Copyright (c) 2026 ITSMBench Contributors
|
|
8
|
+
|
|
9
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
10
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
11
|
+
in the Software without restriction, including without limitation the rights
|
|
12
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
13
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
14
|
+
furnished to do so, subject to the following conditions:
|
|
15
|
+
|
|
16
|
+
The above copyright notice and this permission notice shall be included in all
|
|
17
|
+
copies or substantial portions of the Software.
|
|
18
|
+
|
|
19
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
20
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
21
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
22
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
23
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
24
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
25
|
+
SOFTWARE.
|
|
26
|
+
License-File: LICENSE
|
|
27
|
+
Keywords: audit,benchmark,evaluation,itsm,llm
|
|
28
|
+
Classifier: Development Status :: 3 - Alpha
|
|
29
|
+
Classifier: Intended Audience :: Science/Research
|
|
30
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
31
|
+
Classifier: Programming Language :: Python :: 3
|
|
32
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
33
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
34
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
35
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
36
|
+
Requires-Python: >=3.10
|
|
37
|
+
Description-Content-Type: text/markdown
|
|
38
|
+
|
|
39
|
+
# ITSMBench Audit
|
|
40
|
+
|
|
41
|
+
Static authoring-time audit tooling for [ITSMBench](https://github.com/new-measure/ITSMBench) tasks.
|
|
42
|
+
|
|
43
|
+
ITSMBench evaluates how well AI agents perform realistic IT service-management work. Each task places an agent in a containerised enterprise environment, gives it a ticket-style instruction, and evaluates the resulting environment using a hidden verifier. The benchmark spans areas such as incident management, access management, offboarding, and security response.
|
|
44
|
+
|
|
45
|
+
ITSMBench Audit is a preflight tool for task authors. Before a task is used in a benchmark run, it checks whether the task package is structurally sound, what the verifier actually tests, and whether the documented requirements line up with what the verifier enforces.
|
|
46
|
+
|
|
47
|
+
> **Task authoring → `itsmbench-audit` → review findings → fix → benchmark run**
|
|
48
|
+
|
|
49
|
+
---
|
|
50
|
+
|
|
51
|
+
## Why this exists
|
|
52
|
+
|
|
53
|
+
A task can look correct while hiding real problems in its evaluation layer. A required file might be missing. The verifier might test things that are never mentioned in the task description. A documented requirement might have no corresponding verifier check. The verifier might be too complex to reason about statically. An assertion might confirm a final state without proving that the agent actually performed the expected action.
|
|
54
|
+
|
|
55
|
+
These problems matter because the verifier is what ultimately decides whether an agent succeeded. ITSMBench Audit surfaces them before benchmark execution.
|
|
56
|
+
|
|
57
|
+
---
|
|
58
|
+
|
|
59
|
+
## What it checks
|
|
60
|
+
|
|
61
|
+
### 1. Package health
|
|
62
|
+
|
|
63
|
+
Verifies the basic structure of an ITSMBench task: expected files, directories, and references between local files. The check is straightforward — can this task be plausibly executed and evaluated as written?
|
|
64
|
+
|
|
65
|
+
---
|
|
66
|
+
|
|
67
|
+
### 2. Verifier inventory
|
|
68
|
+
|
|
69
|
+
Reads the verifier and tries to extract what each test is actually asserting. Supported sources are Python test files, JavaScript test files, and JSON-based assertion files.
|
|
70
|
+
|
|
71
|
+
For each assertion, the analyser tries to extract structured information:
|
|
72
|
+
|
|
73
|
+
| Field | Description |
|
|
74
|
+
|---|---|
|
|
75
|
+
| `action` | The operation asserted |
|
|
76
|
+
| `object` | The resource acted upon |
|
|
77
|
+
| `entity` | The subject performing the action |
|
|
78
|
+
| `state` | The resulting or required state |
|
|
79
|
+
| `field` | A specific attribute under inspection |
|
|
80
|
+
| `expected_value` | The value the field must hold |
|
|
81
|
+
| `forbidden_value` | A value the field must not hold |
|
|
82
|
+
| `relation` | A relationship between entities |
|
|
83
|
+
| `predicate` | A general logical predicate |
|
|
84
|
+
| `scope` | The boundary within which the assertion applies |
|
|
85
|
+
| `provenance` | Where the assertion originates |
|
|
86
|
+
|
|
87
|
+
Python verifiers are analysed using static AST inspection and bounded helper analysis — the verifier is never executed. For JavaScript, the analyser recognises common assertion idioms: `assert`, `expect(...).toBe(...)`, `toEqual(...)`, `toContain(...)`, and related forms. JSON assertions are normalised into the same internal representation.
|
|
88
|
+
|
|
89
|
+
---
|
|
90
|
+
|
|
91
|
+
### 3. Documentation and verifier consistency
|
|
92
|
+
|
|
93
|
+
ITSMBench tasks describe what the agent should do in natural language. The audit compares those descriptions against the semantics extracted from the verifier, running the comparison in both directions.
|
|
94
|
+
|
|
95
|
+
First, it checks whether the verifier actually tests what the task description says to do. Then it checks the other way: whether the verifier enforces things that are never mentioned in the description.
|
|
96
|
+
|
|
97
|
+
The matching is conservative. A weak or ambiguous relationship is flagged for review rather than silently counted as a pass.
|
|
98
|
+
|
|
99
|
+
---
|
|
100
|
+
|
|
101
|
+
## Static analysis, not execution
|
|
102
|
+
|
|
103
|
+
ITSMBench Audit never runs an agent, task environment, or verifier. It reads source files only.
|
|
104
|
+
|
|
105
|
+
```mermaid
|
|
106
|
+
flowchart TD
|
|
107
|
+
T["ITSMBench task\n─────────────────\ninstruction\nenvironment\nverifier"]
|
|
108
|
+
A["itsmbench-audit"]
|
|
109
|
+
P["Package health"]
|
|
110
|
+
V["Verifier inventory"]
|
|
111
|
+
D["Documentation consistency"]
|
|
112
|
+
R["Review findings"]
|
|
113
|
+
F["Fix task / verifier"]
|
|
114
|
+
B["Run ITSMBench"]
|
|
115
|
+
|
|
116
|
+
T --> A
|
|
117
|
+
A --> P
|
|
118
|
+
A --> V
|
|
119
|
+
A --> D
|
|
120
|
+
P --> R
|
|
121
|
+
V --> R
|
|
122
|
+
D --> R
|
|
123
|
+
R --> F
|
|
124
|
+
F --> B
|
|
125
|
+
```
|
|
126
|
+
|
|
127
|
+
---
|
|
128
|
+
|
|
129
|
+
## Installation
|
|
130
|
+
|
|
131
|
+
Install from PyPI:
|
|
132
|
+
|
|
133
|
+
```sh
|
|
134
|
+
pip install itsmbench-audit
|
|
135
|
+
```
|
|
136
|
+
|
|
137
|
+
Or install the repository locally:
|
|
138
|
+
|
|
139
|
+
```sh
|
|
140
|
+
git clone https://github.com/souro26/ITSMBench-Audit
|
|
141
|
+
cd itsmbench-audit
|
|
142
|
+
pip install .
|
|
143
|
+
```
|
|
144
|
+
|
|
145
|
+
For development:
|
|
146
|
+
|
|
147
|
+
```sh
|
|
148
|
+
pip install -e .
|
|
149
|
+
```
|
|
150
|
+
|
|
151
|
+
---
|
|
152
|
+
|
|
153
|
+
## Usage
|
|
154
|
+
|
|
155
|
+
### Audit a single task
|
|
156
|
+
|
|
157
|
+
```sh
|
|
158
|
+
itsmbench-audit /path/to/ITSMBench/tasks/task-a-1
|
|
159
|
+
```
|
|
160
|
+
|
|
161
|
+
On Windows:
|
|
162
|
+
|
|
163
|
+
```sh
|
|
164
|
+
itsmbench-audit C:\path\to\ITSMBench\tasks\task-a-1
|
|
165
|
+
```
|
|
166
|
+
|
|
167
|
+
You can also invoke the package directly:
|
|
168
|
+
|
|
169
|
+
```sh
|
|
170
|
+
python -m itsmbench_audit /path/to/ITSMBench/tasks/task-a-1
|
|
171
|
+
```
|
|
172
|
+
|
|
173
|
+
### Audit the complete task corpus
|
|
174
|
+
|
|
175
|
+
```sh
|
|
176
|
+
itsmbench-audit --all /path/to/ITSMBench/tasks
|
|
177
|
+
```
|
|
178
|
+
|
|
179
|
+
The corpus mode audits every task and prints aggregate statistics across the full set.
|
|
180
|
+
|
|
181
|
+
---
|
|
182
|
+
|
|
183
|
+
## Understanding the output
|
|
184
|
+
|
|
185
|
+
### `EXTRACTED`
|
|
186
|
+
|
|
187
|
+
The verifier semantics were statically extracted with enough confidence to be useful.
|
|
188
|
+
|
|
189
|
+
```text
|
|
190
|
+
[CHECK:EXTRACTED] test_account_suspended action=suspend entity=user state=suspended
|
|
191
|
+
```
|
|
192
|
+
|
|
193
|
+
This is the strongest result the analyser produces.
|
|
194
|
+
|
|
195
|
+
---
|
|
196
|
+
|
|
197
|
+
### `PARTIAL`
|
|
198
|
+
|
|
199
|
+
The analyser understood part of the assertion but not all of it. This often happens when a verifier delegates logic to a helper function whose return value cannot be fully resolved from the source.
|
|
200
|
+
|
|
201
|
+
```text
|
|
202
|
+
[CHECK:PARTIAL] test_service_configuration ... reason=helper predicate partially resolved
|
|
203
|
+
```
|
|
204
|
+
|
|
205
|
+
`PARTIAL` does not mean the verifier is broken. It means there was not enough static information to complete the extraction.
|
|
206
|
+
|
|
207
|
+
---
|
|
208
|
+
|
|
209
|
+
### `UNEXTRACTED`
|
|
210
|
+
|
|
211
|
+
A verifier test was found, but the assertion semantics could not be extracted at all. These should be reviewed manually.
|
|
212
|
+
|
|
213
|
+
```text
|
|
214
|
+
[CHECK:UNEXTRACTED] test_complex_workflow reason=assertion semantics could not be statically extracted
|
|
215
|
+
```
|
|
216
|
+
|
|
217
|
+
---
|
|
218
|
+
|
|
219
|
+
### `UNMAPPED`
|
|
220
|
+
|
|
221
|
+
The verifier semantics were extracted, but no documented requirement matched strongly enough. This is a documentation gap finding, not necessarily a verifier defect.
|
|
222
|
+
|
|
223
|
+
```text
|
|
224
|
+
[REVIEW] verifier assertion has no documented requirement
|
|
225
|
+
```
|
|
226
|
+
|
|
227
|
+
---
|
|
228
|
+
|
|
229
|
+
### `WEAK`
|
|
230
|
+
|
|
231
|
+
A relationship between a documented requirement and a verifier assertion was found, but the match is not strong enough to count as confirmed. These are reported so authors can decide whether the relationship is intentional.
|
|
232
|
+
|
|
233
|
+
Some action pairs that trigger weak matches:
|
|
234
|
+
|
|
235
|
+
```text
|
|
236
|
+
remove <-> revoke
|
|
237
|
+
delete <-> remove
|
|
238
|
+
suspend <-> deactivate
|
|
239
|
+
```
|
|
240
|
+
|
|
241
|
+
These might be equivalent in one task context and meaningfully different in another. The tool does not make that call for you.
|
|
242
|
+
|
|
243
|
+
---
|
|
244
|
+
|
|
245
|
+
### Actions versus states
|
|
246
|
+
|
|
247
|
+
The audit distinguishes between what an agent did and what state resulted from it.
|
|
248
|
+
|
|
249
|
+
`account.status == "ACTIVE"` tells you the account is active. It does not prove the agent ran a `restore` operation. `account.status == "DEPROVISIONED"` tells you the account was deprovisioned, but not which operation produced that result or whether it was the right one.
|
|
250
|
+
|
|
251
|
+
A verifier that only checks final state is weaker evidence than one that confirms the specific action was taken. The audit surfaces that distinction.
|
|
252
|
+
|
|
253
|
+
---
|
|
254
|
+
|
|
255
|
+
## What the tool does not do
|
|
256
|
+
|
|
257
|
+
ITSMBench Audit reads task source files and reports what it finds. It does not evaluate agents, execute verifiers, judge output quality, or replace any part of the benchmark runtime. It is not an LLM judge, a security scanner, or a symbolic execution engine.
|
|
258
|
+
|
|
259
|
+
It produces static evidence. Whether that evidence is sufficient is a judgement for the task author.
|
|
260
|
+
|
|
261
|
+
---
|
|
262
|
+
|
|
263
|
+
## Conservative by design
|
|
264
|
+
|
|
265
|
+
The analyser is intentionally cautious. When semantics cannot be established from the source, it reports that clearly rather than guessing.
|
|
266
|
+
|
|
267
|
+
Concretely: test function names are treated as weak hints, not proof. Final states are not automatically read as actions. Similar-sounding actions are not assumed to be equivalent. Ambiguous control flow stays ambiguous. Unresolved helper logic stays unresolved. A `PARTIAL` result is reported as partial, not rounded up to `EXTRACTED`.
|
|
268
|
+
|
|
269
|
+
The output is meant to be inspectable, not optimistic.
|
|
270
|
+
|
|
271
|
+
---
|
|
272
|
+
|
|
273
|
+
## Design
|
|
274
|
+
|
|
275
|
+
Each verifier source — Python AST, JavaScript assertions, JSON — is parsed into a common intermediate representation. That IR is then compared against the requirement IR derived from the task description.
|
|
276
|
+
|
|
277
|
+
```mermaid
|
|
278
|
+
flowchart TD
|
|
279
|
+
PY["Python AST"]
|
|
280
|
+
JS["JavaScript assertions"]
|
|
281
|
+
JSON["JSON assertions"]
|
|
282
|
+
IR["VerifierAssertion IR"]
|
|
283
|
+
REQ["Requirement IR"]
|
|
284
|
+
COV["Coverage / matching analysis"]
|
|
285
|
+
RPT["Audit report"]
|
|
286
|
+
|
|
287
|
+
PY --> IR
|
|
288
|
+
JS --> IR
|
|
289
|
+
JSON --> IR
|
|
290
|
+
IR --> REQ
|
|
291
|
+
REQ --> COV
|
|
292
|
+
COV --> RPT
|
|
293
|
+
```
|
|
294
|
+
|
|
295
|
+
Each assertion in the IR can carry:
|
|
296
|
+
|
|
297
|
+
| Field | Description |
|
|
298
|
+
|---|---|
|
|
299
|
+
| `action` | The primary operation |
|
|
300
|
+
| `implied_action` | An action implied by the assertion |
|
|
301
|
+
| `object` | The resource under test |
|
|
302
|
+
| `entity` | The actor |
|
|
303
|
+
| `qualifier` | A modifier on the action or state |
|
|
304
|
+
| `state` | The required or resulting state |
|
|
305
|
+
| `field` | The specific attribute inspected |
|
|
306
|
+
| `expected_value` | The value the attribute must hold |
|
|
307
|
+
| `forbidden_value` | A value the attribute must not hold |
|
|
308
|
+
| `relation` | A relationship between entities |
|
|
309
|
+
| `predicate` | A general logical predicate |
|
|
310
|
+
| `scope` | The assertion boundary |
|
|
311
|
+
| `provenance` | Extraction origin |
|
|
312
|
+
| `raw_evidence` | The original source fragment |
|
|
313
|
+
| `extraction_status` | `EXTRACTED`, `PARTIAL`, or `UNEXTRACTED` |
|
|
314
|
+
|
|
315
|
+
Using a common IR means the same coverage logic applies to all verifier formats.
|
|
316
|
+
|
|
317
|
+
---
|
|
318
|
+
|
|
319
|
+
## Relationship to ITSMBench
|
|
320
|
+
|
|
321
|
+
[ITSMBench](https://github.com/new-measure/ITSMBench) runs AI agents on IT service-management tasks in containerised environments and evaluates them using hidden verifiers. ITSMBench Audit sits one step earlier, during task authoring, before the benchmark is run.
|
|
322
|
+
|
|
323
|
+
```mermaid
|
|
324
|
+
flowchart TD
|
|
325
|
+
AUTH["Task authoring"]
|
|
326
|
+
TASK["ITSMBench task"]
|
|
327
|
+
AUDIT["itsmbench-audit"]
|
|
328
|
+
FIX["Review & fix"]
|
|
329
|
+
RUN["ITSMBench run"]
|
|
330
|
+
EVAL["Agent evaluation"]
|
|
331
|
+
|
|
332
|
+
AUTH --> TASK
|
|
333
|
+
TASK --> AUDIT
|
|
334
|
+
AUDIT --> FIX
|
|
335
|
+
FIX --> RUN
|
|
336
|
+
RUN --> EVAL
|
|
337
|
+
```
|
|
338
|
+
|
|
339
|
+
The benchmark supports both remote Daytona execution and local Docker execution. ITSMBench Audit fits into the local development part of that workflow, before anything is actually run.
|
|
340
|
+
|
|
341
|
+
---
|
|
342
|
+
|
|
343
|
+
## Limitations
|
|
344
|
+
|
|
345
|
+
Some verifier logic cannot be recovered statically. This includes dynamic dispatch, complex control flow, deeply nested helpers, external API calls, dynamically constructed data, and logic that only becomes clear at runtime.
|
|
346
|
+
|
|
347
|
+
When the analyser hits these cases it reports `PARTIAL` or `UNEXTRACTED`. That is the correct output. Surfacing uncertainty is more useful than producing a confident-sounding result that is not actually grounded in the source.
|
|
348
|
+
|
|
349
|
+
Passing the audit also does not mean the verifier is correct. It means the documentation and verifier appear consistent to a static analyser. That is one signal among several when assessing task quality.
|
|
350
|
+
|
|
351
|
+
---
|
|
352
|
+
|
|
353
|
+
## Development
|
|
354
|
+
|
|
355
|
+
```sh
|
|
356
|
+
git clone https://github.com/souro26/ITSMBench-Audit
|
|
357
|
+
cd itsmbench-audit
|
|
358
|
+
pip install -e .
|
|
359
|
+
```
|
|
360
|
+
|
|
361
|
+
Run a syntax check:
|
|
362
|
+
|
|
363
|
+
```sh
|
|
364
|
+
python -m py_compile itsmbench_audit/__main__.py
|
|
365
|
+
```
|
|
366
|
+
|
|
367
|
+
Run the audit against a task:
|
|
368
|
+
|
|
369
|
+
```sh
|
|
370
|
+
python -m itsmbench_audit /path/to/ITSMBench/tasks/task-a-1
|
|
371
|
+
```
|
|
372
|
+
|
|
373
|
+
Run the full corpus:
|
|
374
|
+
|
|
375
|
+
```sh
|
|
376
|
+
python -m itsmbench_audit --all /path/to/ITSMBench/tasks
|
|
377
|
+
```
|
|
378
|
+
|
|
379
|
+
---
|
|
380
|
+
|
|
381
|
+
## License
|
|
382
|
+
|
|
383
|
+
This project is licensed under the [MIT License](LICENSE).
|
|
384
|
+
|