ww-agentic-workflows 1.0.0.dev3__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ww/__init__.py +18 -0
- ww/_bundled_extensions/ww/git/extension.py +1728 -0
- ww/action_execution.py +887 -0
- ww/actions/__init__.py +94 -0
- ww/actions/command.py +444 -0
- ww/actions/contracts.py +699 -0
- ww/actions/extension.py +197 -0
- ww/actions/mcp.py +84 -0
- ww/actions/prompt.py +74 -0
- ww/actions/skill.py +62 -0
- ww/actions/slash_command.py +63 -0
- ww/agents.py +151 -0
- ww/amendments.py +54 -0
- ww/artifacts.py +93 -0
- ww/assessments.py +181 -0
- ww/assets/__init__.py +2 -0
- ww/assets/agent_instructions.md +49 -0
- ww/assets/docs/examples.md +879 -0
- ww/assets/docs/features.md +4639 -0
- ww/assets/docs/specification.md +1876 -0
- ww/assets/noww_skill.md +11 -0
- ww/assets/workflows/catchall.yaml +26 -0
- ww/assets/workflows/onboarding.yaml +586 -0
- ww/assets/workflows/scriptize.yaml +130 -0
- ww/assets/ww-automate_skill.md +23 -0
- ww/assets/ww-deduce-feedback_skill.md +38 -0
- ww/assets/ww-feedback-rules_skill.md +48 -0
- ww/assets/ww-learn-project_skill.md +22 -0
- ww/assets/ww-refresh_skill.md +26 -0
- ww/assets/ww-rule_skill.md +83 -0
- ww/assets/ww-rules-from-artifacts_skill.md +22 -0
- ww/assets/ww-scriptize_skill.md +33 -0
- ww/assets/ww-setup_skill.md +94 -0
- ww/assets/ww-solve_skill.md +23 -0
- ww/assets/ww-suggest_skill.md +32 -0
- ww/assets/ww-wizard_skill.md +105 -0
- ww/assets/ww_skill.md +59 -0
- ww/assignments.py +283 -0
- ww/bootstrap.py +405 -0
- ww/builtin_workflows.py +215 -0
- ww/changes.py +225 -0
- ww/child_coordination.py +482 -0
- ww/children.py +106 -0
- ww/claude_permissions.py +115 -0
- ww/cli/__init__.py +7 -0
- ww/cli/__main__.py +6 -0
- ww/cli/audit.py +129 -0
- ww/cli/catalogs.py +131 -0
- ww/cli/discover.py +607 -0
- ww/cli/initialization.py +898 -0
- ww/cli/lookup.py +287 -0
- ww/cli/main.py +1768 -0
- ww/cli/parser.py +1200 -0
- ww/cli/prompts.py +217 -0
- ww/cli/updates.py +117 -0
- ww/completion_artifacts.py +156 -0
- ww/completion_inputs.py +39 -0
- ww/config/__init__.py +582 -0
- ww/config/actions.py +591 -0
- ww/config/composition.py +571 -0
- ww/config/rules.py +511 -0
- ww/config/steps.py +1220 -0
- ww/config/values.py +223 -0
- ww/config_files.py +191 -0
- ww/config_writes.py +264 -0
- ww/contracts.py +155 -0
- ww/control.py +41 -0
- ww/defaults.py +130 -0
- ww/design_docs.py +32 -0
- ww/discovery.py +104 -0
- ww/documents.py +217 -0
- ww/errors.py +18 -0
- ww/executable.py +43 -0
- ww/execution_models/__init__.py +64 -0
- ww/execution_models/construction.py +148 -0
- ww/execution_models/decoding.py +38 -0
- ww/execution_models/plan_codec.py +565 -0
- ww/execution_models/records.py +1206 -0
- ww/execution_models/runs.py +266 -0
- ww/extensions/__init__.py +40 -0
- ww/extensions/api.py +559 -0
- ww/extensions/registry.py +864 -0
- ww/extensions/store.py +78 -0
- ww/feedback.py +342 -0
- ww/handler_repairs.py +57 -0
- ww/hooks/__init__.py +40 -0
- ww/hooks/agents.py +380 -0
- ww/hooks/install.py +168 -0
- ww/hooks/notices.py +206 -0
- ww/hooks/records.py +209 -0
- ww/hooks/runtime.py +266 -0
- ww/hooks/transcripts.py +183 -0
- ww/inspect.py +896 -0
- ww/instructions/__init__.py +17 -0
- ww/instructions/builder.py +1682 -0
- ww/instructions/commands.py +335 -0
- ww/instructions/handoff.py +149 -0
- ww/instructions/models.py +686 -0
- ww/instructions/policy.py +219 -0
- ww/instructions/text.py +168 -0
- ww/interactions.py +187 -0
- ww/interpolation.py +37 -0
- ww/item_passes.py +167 -0
- ww/items.py +99 -0
- ww/locking.py +207 -0
- ww/metadata_publication.py +230 -0
- ww/onboarding.py +229 -0
- ww/open_work.py +236 -0
- ww/operations.py +193 -0
- ww/operator_ui/__init__.py +16 -0
- ww/operator_ui/page.html +351 -0
- ww/operator_ui/server.py +215 -0
- ww/operator_ui/session.py +389 -0
- ww/operator_ui/sheet.py +104 -0
- ww/operator_ui/view.py +109 -0
- ww/output.py +339 -0
- ww/output_adapters/__init__.py +12 -0
- ww/output_adapters/base.py +25 -0
- ww/output_adapters/json_adapter.py +37 -0
- ww/output_adapters/markdown.py +2293 -0
- ww/output_adapters/rule_pages.py +337 -0
- ww/output_adapters/terminal.py +21 -0
- ww/package_updates.py +167 -0
- ww/plan/__init__.py +38 -0
- ww/plan/actions.py +207 -0
- ww/plan/compiler.py +1492 -0
- ww/plan/constructs.py +456 -0
- ww/plan/models.py +665 -0
- ww/project_config.py +752 -0
- ww/recovery.py +401 -0
- ww/replanning.py +367 -0
- ww/results.py +77 -0
- ww/rule_checks.py +230 -0
- ww/rule_conversion.py +331 -0
- ww/rule_disputes.py +148 -0
- ww/rule_store.py +456 -0
- ww/rule_verification.py +714 -0
- ww/rule_views.py +447 -0
- ww/rule_writes.py +920 -0
- ww/run_coordination.py +158 -0
- ww/runtimes.py +105 -0
- ww/service.py +4405 -0
- ww/setup_apply.py +428 -0
- ww/step_values.py +20 -0
- ww/storage.py +447 -0
- ww/storage_adapters/__init__.py +36 -0
- ww/storage_adapters/base.py +540 -0
- ww/storage_adapters/filesystem.py +370 -0
- ww/storage_adapters/memory.py +195 -0
- ww/storage_adapters/project_metadata.py +69 -0
- ww/storage_adapters/task_document.py +484 -0
- ww/task_ids.py +114 -0
- ww/task_references.py +124 -0
- ww/transitions.py +1619 -0
- ww/updates.py +399 -0
- ww/upgrade.py +95 -0
- ww/validation.py +168 -0
- ww/variables.py +275 -0
- ww/workflow_config.py +854 -0
- ww/workflow_update.py +239 -0
- ww/workflow_validation.py +1260 -0
- ww/workspace.py +50 -0
- ww_agentic_workflows-1.0.0.dev3.dist-info/METADATA +690 -0
- ww_agentic_workflows-1.0.0.dev3.dist-info/RECORD +167 -0
- ww_agentic_workflows-1.0.0.dev3.dist-info/WHEEL +4 -0
- ww_agentic_workflows-1.0.0.dev3.dist-info/entry_points.txt +2 -0
- ww_agentic_workflows-1.0.0.dev3.dist-info/licenses/LICENSE +674 -0
ww/rule_verification.py
ADDED
|
@@ -0,0 +1,714 @@
|
|
|
1
|
+
# SPDX-License-Identifier: GPL-3.0-or-later
|
|
2
|
+
"""Verification of rules without a command.
|
|
3
|
+
|
|
4
|
+
A rule without a command is never graded by the worker who did the step. As
|
|
5
|
+
the step begins, each such rule resolves against the rule-automation store:
|
|
6
|
+
a rule whose wording has a converted check is checked by it, where the
|
|
7
|
+
check's configuration files exist; every other rule is judged. When the
|
|
8
|
+
step's worker completes and its checks pass, ww holds the completion and
|
|
9
|
+
inserts verification items right before the step: ww-generated agent items,
|
|
10
|
+
one per distinct worker hint set among the judged rules, each giving a
|
|
11
|
+
``pass`` or ``fail`` verdict on its rules. A verifier never writes the store:
|
|
12
|
+
turning rules into checks is ``ww-scriptize-rules``'s job, outside tasks.
|
|
13
|
+
|
|
14
|
+
A failing verdict sends the step back to its worker through the fix loop;
|
|
15
|
+
once every rule passes, ww records the held completion.
|
|
16
|
+
|
|
17
|
+
This module holds the pure parts: resolving a step's rules against the store,
|
|
18
|
+
choosing what a round must ask, the verification items and round records,
|
|
19
|
+
and parsing verifier results. The service orchestrates them inside
|
|
20
|
+
``complete`` and ``next``.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
from __future__ import annotations
|
|
24
|
+
|
|
25
|
+
import json
|
|
26
|
+
from dataclasses import dataclass, replace
|
|
27
|
+
from pathlib import Path
|
|
28
|
+
from typing import Any
|
|
29
|
+
|
|
30
|
+
from ww.actions import PlannedAction, Prompt, actions
|
|
31
|
+
from ww.contracts import Verdict
|
|
32
|
+
from ww.errors import StateError
|
|
33
|
+
from ww.execution_models import (
|
|
34
|
+
CheckReport,
|
|
35
|
+
CheckResult,
|
|
36
|
+
ExecutionState,
|
|
37
|
+
PlanSnapshot,
|
|
38
|
+
build_step_projection,
|
|
39
|
+
new_item_execution,
|
|
40
|
+
operation_scope_for,
|
|
41
|
+
)
|
|
42
|
+
from ww.execution_models.records import (
|
|
43
|
+
HeldCompletion,
|
|
44
|
+
PlanItemExecution,
|
|
45
|
+
RuleResolution,
|
|
46
|
+
RuleVerdict,
|
|
47
|
+
VerificationRule,
|
|
48
|
+
)
|
|
49
|
+
from ww.plan import (
|
|
50
|
+
PlanItem,
|
|
51
|
+
PlannedCheck,
|
|
52
|
+
PlannedRule,
|
|
53
|
+
VerificationTarget,
|
|
54
|
+
WorkflowPlan,
|
|
55
|
+
number_step_paths,
|
|
56
|
+
)
|
|
57
|
+
from ww.rule_store import CheckEntry, CheckSpec, RuleAutomation, is_config_path
|
|
58
|
+
from ww.transitions import Clock, project_steps
|
|
59
|
+
from ww.workflow_config import RuleHints
|
|
60
|
+
|
|
61
|
+
SKIPPED_ROUND = "skipped: nothing to verify in this round"
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def judged_rules(item: PlanItem) -> tuple[PlannedRule, ...]:
|
|
65
|
+
"""The step's rules without a command of their own."""
|
|
66
|
+
return tuple(rule for rule in item.rules if not rule.has_command)
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def to_verify(item: PlanItem, record: PlanItemExecution) -> tuple[PlannedRule, ...]:
|
|
70
|
+
"""The step's rules without a command that the operator did not waive."""
|
|
71
|
+
waived = {key for key, _ in record.checks_waived}
|
|
72
|
+
return tuple(rule for rule in judged_rules(item) if rule.id not in waived)
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def resolve_rules(
|
|
76
|
+
item: PlanItem,
|
|
77
|
+
automation: RuleAutomation,
|
|
78
|
+
*,
|
|
79
|
+
directory: Path | None = None,
|
|
80
|
+
) -> tuple[tuple[RuleResolution, ...], tuple[PlannedCheck, ...]]:
|
|
81
|
+
"""How each rule without a command is enforced, as the step begins.
|
|
82
|
+
|
|
83
|
+
A rule whose wording has a converted check is checked by it; several
|
|
84
|
+
rules sharing one check get one planned check that covers them all.
|
|
85
|
+
Every other rule is judged by a verifier, whatever else the store says
|
|
86
|
+
about it. With ``directory``, the one the step's checks run in, a
|
|
87
|
+
converted check whose configuration files are not all there, such as one
|
|
88
|
+
built on a branch not merged yet, does not apply: its rules are judged,
|
|
89
|
+
naming the missing file.
|
|
90
|
+
"""
|
|
91
|
+
resolutions: list[RuleResolution] = []
|
|
92
|
+
covered: dict[str, list[PlannedRule]] = {}
|
|
93
|
+
specs: dict[str, CheckEntry] = {}
|
|
94
|
+
for rule in judged_rules(item):
|
|
95
|
+
entry = automation.rules.get(rule.text_hash)
|
|
96
|
+
interpretation = entry.interpretation if entry else None
|
|
97
|
+
converted = automation.converted_check(rule.text_hash)
|
|
98
|
+
if converted is not None:
|
|
99
|
+
name, check = converted
|
|
100
|
+
missing = _missing_config(check.spec, directory)
|
|
101
|
+
if missing is not None:
|
|
102
|
+
resolutions.append(
|
|
103
|
+
RuleResolution(
|
|
104
|
+
id=rule.id,
|
|
105
|
+
status="judged",
|
|
106
|
+
check=name,
|
|
107
|
+
interpretation=interpretation,
|
|
108
|
+
missing=missing,
|
|
109
|
+
)
|
|
110
|
+
)
|
|
111
|
+
continue
|
|
112
|
+
covered.setdefault(name, []).append(rule)
|
|
113
|
+
specs[name] = check
|
|
114
|
+
resolutions.append(
|
|
115
|
+
RuleResolution(rule.id, "converted", name, interpretation)
|
|
116
|
+
)
|
|
117
|
+
continue
|
|
118
|
+
resolutions.append(RuleResolution(rule.id, "judged", None, interpretation))
|
|
119
|
+
checks = tuple(
|
|
120
|
+
derived_check(name, specs[name].spec, rules) for name, rules in covered.items()
|
|
121
|
+
)
|
|
122
|
+
return tuple(resolutions), checks
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def _missing_config(spec: CheckSpec, directory: Path | None) -> str | None:
|
|
126
|
+
"""The first of a check's configuration files ``directory`` lacks.
|
|
127
|
+
|
|
128
|
+
A path that would reach outside ``directory`` (absolute, or with a ``..``
|
|
129
|
+
part, as only a hand-edited store holds) counts as missing.
|
|
130
|
+
"""
|
|
131
|
+
if directory is None:
|
|
132
|
+
return None
|
|
133
|
+
return next(
|
|
134
|
+
(
|
|
135
|
+
path
|
|
136
|
+
for path in spec.config
|
|
137
|
+
if not is_config_path(path) or not (directory / path).exists()
|
|
138
|
+
),
|
|
139
|
+
None,
|
|
140
|
+
)
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def derived_check(name: str, spec: CheckSpec, rules: list[PlannedRule]) -> PlannedCheck:
|
|
144
|
+
"""One converted store check, planned for the step rules it covers.
|
|
145
|
+
|
|
146
|
+
Its globs are the union of its rules' globs, or none when any covered
|
|
147
|
+
rule applies to every file; it may fail as often as the most lenient of
|
|
148
|
+
its rules allows.
|
|
149
|
+
"""
|
|
150
|
+
paths: tuple[str, ...] = (
|
|
151
|
+
()
|
|
152
|
+
if any(not rule.paths for rule in rules)
|
|
153
|
+
else tuple(dict.fromkeys(path for rule in rules for path in rule.paths))
|
|
154
|
+
)
|
|
155
|
+
return PlannedCheck(
|
|
156
|
+
id=name,
|
|
157
|
+
source="derived",
|
|
158
|
+
summary=f"check {name}",
|
|
159
|
+
command=spec.command,
|
|
160
|
+
paths=paths,
|
|
161
|
+
max_fixes=max(rule.max_fixes for rule in rules),
|
|
162
|
+
covers=tuple(rule.id for rule in rules),
|
|
163
|
+
)
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def verification_needs(
|
|
167
|
+
item: PlanItem, record: PlanItemExecution
|
|
168
|
+
) -> tuple[VerificationRule, ...]:
|
|
169
|
+
"""The rules of a completing step that a verifier must still judge.
|
|
170
|
+
|
|
171
|
+
Rules checked by a resolved derived check, rules with a verdict in the
|
|
172
|
+
current hold, and rules the operator waived for this step are done for
|
|
173
|
+
now. What the step began with says the rest: a rule's reading, and the
|
|
174
|
+
converted check whose configuration is missing here.
|
|
175
|
+
"""
|
|
176
|
+
covered = {rule_id for check in record.resolved_checks for rule_id in check.covers}
|
|
177
|
+
covered.update(key for key, _ in record.checks_waived)
|
|
178
|
+
held = record.held_completion
|
|
179
|
+
began = {resolution.id: resolution for resolution in record.rule_resolutions}
|
|
180
|
+
needs: list[VerificationRule] = []
|
|
181
|
+
for rule in judged_rules(item):
|
|
182
|
+
if rule.id in covered or (held is not None and held.verdict(rule.id)):
|
|
183
|
+
continue
|
|
184
|
+
resolution = began.get(rule.id)
|
|
185
|
+
needs.append(
|
|
186
|
+
VerificationRule(
|
|
187
|
+
id=rule.id,
|
|
188
|
+
text=rule.text,
|
|
189
|
+
text_hash=rule.text_hash,
|
|
190
|
+
interpretation=resolution.interpretation if resolution else None,
|
|
191
|
+
check=resolution.check if resolution and resolution.missing else None,
|
|
192
|
+
missing=resolution.missing if resolution else None,
|
|
193
|
+
)
|
|
194
|
+
)
|
|
195
|
+
return tuple(needs)
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
def effective_hints(rule: PlannedRule, item: PlanItem) -> RuleHints:
|
|
199
|
+
"""The worker a rule is verified by: the rule's hints, else the step's."""
|
|
200
|
+
return RuleHints(
|
|
201
|
+
rule.hints.agent or item.requested_agent,
|
|
202
|
+
rule.hints.model or item.requested_model,
|
|
203
|
+
rule.hints.reasoning or item.requested_reasoning,
|
|
204
|
+
)
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
def verification_item(item: PlanItem, ordinal: int, hints: RuleHints) -> PlanItem:
|
|
208
|
+
"""A ww-generated agent item that verifies ``item``'s rules for ``hints``.
|
|
209
|
+
|
|
210
|
+
It sits in the step's ``before_complete`` phase, which is when it runs:
|
|
211
|
+
the step's completion is held until it is done. It carries nothing of the
|
|
212
|
+
step's own work, and always allows a subagent, so the verifier is never
|
|
213
|
+
the worker who did the step unless one session performs every step.
|
|
214
|
+
"""
|
|
215
|
+
text = (
|
|
216
|
+
f"Verify the rules of the `{item.name}` step: judge its work from the "
|
|
217
|
+
"change set and the rules alone, and do not change it. The "
|
|
218
|
+
"Verification section lists the rules, the change set, and what to "
|
|
219
|
+
"report for each rule."
|
|
220
|
+
)
|
|
221
|
+
prompt = actions.get("prompt")
|
|
222
|
+
|
|
223
|
+
def concrete(value: str | None, fallback: str | None) -> str | None:
|
|
224
|
+
return value if value is not None and value != "auto" else fallback
|
|
225
|
+
|
|
226
|
+
return replace(
|
|
227
|
+
item,
|
|
228
|
+
id=f"{item.workflow}:{item.step}:verify:{ordinal}",
|
|
229
|
+
name=f"{item.name}-verify-{ordinal}",
|
|
230
|
+
description=text,
|
|
231
|
+
operation=PlannedAction("prompt", Prompt(text)),
|
|
232
|
+
owner=prompt.owner,
|
|
233
|
+
execution=prompt.execution,
|
|
234
|
+
requires_agent_input=False,
|
|
235
|
+
phase="before_complete",
|
|
236
|
+
source="internal",
|
|
237
|
+
registered_handler=None,
|
|
238
|
+
provide=(),
|
|
239
|
+
save_metadata=(),
|
|
240
|
+
update_document=(),
|
|
241
|
+
update_item=(),
|
|
242
|
+
outputs=(),
|
|
243
|
+
dependencies=(),
|
|
244
|
+
requested_agent=hints.agent,
|
|
245
|
+
requested_model=hints.model,
|
|
246
|
+
requested_reasoning=hints.reasoning,
|
|
247
|
+
role="worker",
|
|
248
|
+
interactive=False,
|
|
249
|
+
choices=(),
|
|
250
|
+
ui=False,
|
|
251
|
+
model=concrete(hints.model, item.model),
|
|
252
|
+
reasoning=concrete(hints.reasoning, item.reasoning),
|
|
253
|
+
profile=None,
|
|
254
|
+
profile_instruction=None,
|
|
255
|
+
profile_path=None,
|
|
256
|
+
summary=False,
|
|
257
|
+
item_operation=None,
|
|
258
|
+
item_template=False,
|
|
259
|
+
item_collect_only=False,
|
|
260
|
+
shared_items=False,
|
|
261
|
+
item_identity=None,
|
|
262
|
+
item_unique=(),
|
|
263
|
+
split_instruction=None,
|
|
264
|
+
artifact=True,
|
|
265
|
+
child_operation=None,
|
|
266
|
+
child_identity=False,
|
|
267
|
+
artifact_dependency=None,
|
|
268
|
+
loop_break=None,
|
|
269
|
+
loop_continue=None,
|
|
270
|
+
assessment_question=None,
|
|
271
|
+
assessment_outcomes=(),
|
|
272
|
+
assessment_stops=(),
|
|
273
|
+
assessment_parent=None,
|
|
274
|
+
assessment_outcome=None,
|
|
275
|
+
rules=(),
|
|
276
|
+
checks=(),
|
|
277
|
+
modes=(),
|
|
278
|
+
verifies=VerificationTarget(item.id, ordinal, hints),
|
|
279
|
+
)
|
|
280
|
+
|
|
281
|
+
|
|
282
|
+
def verification_items(plan: WorkflowPlan, item_id: str) -> tuple[int, ...]:
|
|
283
|
+
"""The plan indexes of the verification items of one agent item."""
|
|
284
|
+
return tuple(
|
|
285
|
+
index
|
|
286
|
+
for index, entry in enumerate(plan.items)
|
|
287
|
+
if entry.verifies is not None and entry.verifies.item_id == item_id
|
|
288
|
+
)
|
|
289
|
+
|
|
290
|
+
|
|
291
|
+
def index_of(plan: WorkflowPlan, item_id: str) -> int:
|
|
292
|
+
for index, entry in enumerate(plan.items):
|
|
293
|
+
if entry.id == item_id:
|
|
294
|
+
return index
|
|
295
|
+
raise StateError(f"plan item {item_id!r} is not in the task snapshot")
|
|
296
|
+
|
|
297
|
+
|
|
298
|
+
def hold_completion(
|
|
299
|
+
state: ExecutionState,
|
|
300
|
+
plan: WorkflowPlan,
|
|
301
|
+
held: HeldCompletion,
|
|
302
|
+
artifact: str | None,
|
|
303
|
+
now: Clock,
|
|
304
|
+
) -> ExecutionState:
|
|
305
|
+
"""Keep the active item's completion, unrecorded, until its rules are verified.
|
|
306
|
+
|
|
307
|
+
Verdicts of an earlier round of the same hold are kept. The worker's
|
|
308
|
+
assignment ends: the verifiers are other workers.
|
|
309
|
+
"""
|
|
310
|
+
records = list(state.item_executions)
|
|
311
|
+
record = records[state.cursor]
|
|
312
|
+
earlier = record.held_completion
|
|
313
|
+
records[state.cursor] = replace(
|
|
314
|
+
record,
|
|
315
|
+
held_completion=(
|
|
316
|
+
replace(held, verdicts=earlier.verdicts) if earlier is not None else held
|
|
317
|
+
),
|
|
318
|
+
draft_artifact=artifact,
|
|
319
|
+
status="pending",
|
|
320
|
+
)
|
|
321
|
+
return project_steps(
|
|
322
|
+
_without_assignment(
|
|
323
|
+
replace(
|
|
324
|
+
state,
|
|
325
|
+
status="pending",
|
|
326
|
+
active_item_id=None,
|
|
327
|
+
item_executions=tuple(records),
|
|
328
|
+
updated_at=now(),
|
|
329
|
+
)
|
|
330
|
+
),
|
|
331
|
+
plan,
|
|
332
|
+
now,
|
|
333
|
+
)
|
|
334
|
+
|
|
335
|
+
|
|
336
|
+
def open_verification_round(
|
|
337
|
+
state: ExecutionState,
|
|
338
|
+
snapshot: PlanSnapshot,
|
|
339
|
+
item: PlanItem,
|
|
340
|
+
needs: tuple[VerificationRule, ...],
|
|
341
|
+
now: Clock,
|
|
342
|
+
) -> tuple[ExecutionState, PlanSnapshot]:
|
|
343
|
+
"""Give each hint set of ``needs`` a verification item and start the round.
|
|
344
|
+
|
|
345
|
+
A hint set without an item yet gets one, inserted right before the step
|
|
346
|
+
as a new plan revision; items from earlier rounds are reused, their
|
|
347
|
+
previous records kept in the history. An item with nothing to ask this
|
|
348
|
+
round is recorded as skipped. The cursor moves to the first item asked.
|
|
349
|
+
"""
|
|
350
|
+
rules = {rule.id: rule for rule in item.rules}
|
|
351
|
+
groups: dict[RuleHints, list[VerificationRule]] = {}
|
|
352
|
+
for need in needs:
|
|
353
|
+
groups.setdefault(effective_hints(rules[need.id], item), []).append(need)
|
|
354
|
+
plan = snapshot.plan
|
|
355
|
+
existing = [plan.items[index] for index in verification_items(plan, item.id)]
|
|
356
|
+
known = {entry.verifies.hints for entry in existing if entry.verifies is not None}
|
|
357
|
+
added = [
|
|
358
|
+
verification_item(item, len(existing) + number, hints)
|
|
359
|
+
for number, hints in enumerate(
|
|
360
|
+
(hints for hints in groups if hints not in known), 1
|
|
361
|
+
)
|
|
362
|
+
]
|
|
363
|
+
if added:
|
|
364
|
+
at = index_of(plan, item.id)
|
|
365
|
+
items = (*plan.items[:at], *added, *plan.items[at:])
|
|
366
|
+
numbered = tuple(
|
|
367
|
+
replace(entry, position=index) for index, entry in enumerate(items, 1)
|
|
368
|
+
)
|
|
369
|
+
plan = replace(plan, items=number_step_paths(numbered))
|
|
370
|
+
previous = {record.plan_item_id: record for record in state.item_executions}
|
|
371
|
+
scope = operation_scope_for(state)
|
|
372
|
+
snapshot = replace(
|
|
373
|
+
snapshot, plan=plan, plan_revision=snapshot.plan_revision + 1
|
|
374
|
+
)
|
|
375
|
+
state = replace(
|
|
376
|
+
state,
|
|
377
|
+
item_executions=tuple(
|
|
378
|
+
replace(previous[entry.id], position=entry.position)
|
|
379
|
+
if entry.id in previous
|
|
380
|
+
else new_item_execution(state.task_id, scope, entry)
|
|
381
|
+
for entry in plan.items
|
|
382
|
+
),
|
|
383
|
+
steps=build_step_projection(plan, state.steps),
|
|
384
|
+
plan_revision=snapshot.plan_revision,
|
|
385
|
+
plan_digest=snapshot.plan_digest,
|
|
386
|
+
)
|
|
387
|
+
records = list(state.item_executions)
|
|
388
|
+
history = list(state.execution_history)
|
|
389
|
+
scope = operation_scope_for(state)
|
|
390
|
+
first: int | None = None
|
|
391
|
+
for index in verification_items(plan, item.id):
|
|
392
|
+
entry = plan.items[index]
|
|
393
|
+
assert entry.verifies is not None
|
|
394
|
+
earlier = records[index]
|
|
395
|
+
if earlier.status != "pending" or earlier.verification:
|
|
396
|
+
history.append(earlier)
|
|
397
|
+
fresh = new_item_execution(state.task_id, scope, entry)
|
|
398
|
+
asked = groups.get(entry.verifies.hints)
|
|
399
|
+
if asked:
|
|
400
|
+
records[index] = replace(fresh, verification=tuple(asked))
|
|
401
|
+
first = index if first is None else first
|
|
402
|
+
else:
|
|
403
|
+
records[index] = replace(
|
|
404
|
+
fresh, status="completed", completed_at=now(), result=SKIPPED_ROUND
|
|
405
|
+
)
|
|
406
|
+
if first is None: # pragma: no cover - callers open a round only with needs
|
|
407
|
+
raise StateError("a verification round needs at least one rule")
|
|
408
|
+
return (
|
|
409
|
+
project_steps(
|
|
410
|
+
_without_assignment(
|
|
411
|
+
replace(
|
|
412
|
+
state,
|
|
413
|
+
status="pending",
|
|
414
|
+
cursor=first,
|
|
415
|
+
active_item_id=None,
|
|
416
|
+
item_executions=tuple(records),
|
|
417
|
+
execution_history=tuple(history),
|
|
418
|
+
updated_at=now(),
|
|
419
|
+
)
|
|
420
|
+
),
|
|
421
|
+
plan,
|
|
422
|
+
now,
|
|
423
|
+
),
|
|
424
|
+
snapshot,
|
|
425
|
+
)
|
|
426
|
+
|
|
427
|
+
|
|
428
|
+
def round_open(state: ExecutionState, plan: WorkflowPlan, item_id: str) -> bool:
|
|
429
|
+
"""Whether a verification item of ``item_id`` is still to be performed."""
|
|
430
|
+
return any(
|
|
431
|
+
state.item_executions[index].status in {"pending", "in_progress"}
|
|
432
|
+
and state.item_executions[index].verification
|
|
433
|
+
for index in verification_items(plan, item_id)
|
|
434
|
+
)
|
|
435
|
+
|
|
436
|
+
|
|
437
|
+
def close_round(
|
|
438
|
+
state: ExecutionState, plan: WorkflowPlan, item_id: str, now: Clock
|
|
439
|
+
) -> ExecutionState:
|
|
440
|
+
"""End a round early: its verifiers not yet performed are skipped.
|
|
441
|
+
|
|
442
|
+
A failing verdict sends the step back to its worker, so what the others
|
|
443
|
+
would judge is judged again on the revised work.
|
|
444
|
+
"""
|
|
445
|
+
records = list(state.item_executions)
|
|
446
|
+
for index in verification_items(plan, item_id):
|
|
447
|
+
if records[index].status == "pending" and records[index].verification:
|
|
448
|
+
records[index] = replace(
|
|
449
|
+
records[index],
|
|
450
|
+
status="completed",
|
|
451
|
+
completed_at=now(),
|
|
452
|
+
result="skipped: an earlier verdict of the round failed",
|
|
453
|
+
)
|
|
454
|
+
return replace(state, item_executions=tuple(records), updated_at=now())
|
|
455
|
+
|
|
456
|
+
|
|
457
|
+
def record_round(
|
|
458
|
+
state: ExecutionState,
|
|
459
|
+
index: int,
|
|
460
|
+
verdicts: tuple[RuleVerdict, ...],
|
|
461
|
+
now: Clock,
|
|
462
|
+
) -> ExecutionState:
|
|
463
|
+
"""Add one verifier's verdicts to the held step's record."""
|
|
464
|
+
records = list(state.item_executions)
|
|
465
|
+
record = records[index]
|
|
466
|
+
held = record.held_completion
|
|
467
|
+
if held is None:
|
|
468
|
+
raise StateError("the verified step has no held completion")
|
|
469
|
+
replaced = {verdict.id for verdict in verdicts}
|
|
470
|
+
records[index] = replace(
|
|
471
|
+
record,
|
|
472
|
+
held_completion=replace(
|
|
473
|
+
held,
|
|
474
|
+
verdicts=(
|
|
475
|
+
*(entry for entry in held.verdicts if entry.id not in replaced),
|
|
476
|
+
*verdicts,
|
|
477
|
+
),
|
|
478
|
+
),
|
|
479
|
+
)
|
|
480
|
+
return replace(state, item_executions=tuple(records), updated_at=now())
|
|
481
|
+
|
|
482
|
+
|
|
483
|
+
def resume_held(
|
|
484
|
+
state: ExecutionState, plan: WorkflowPlan, index: int, now: Clock
|
|
485
|
+
) -> ExecutionState:
|
|
486
|
+
"""Reopen the held step so ww can record its completion as submitted."""
|
|
487
|
+
records = list(state.item_executions)
|
|
488
|
+
records[index] = replace(records[index], status="in_progress", error=None)
|
|
489
|
+
return project_steps(
|
|
490
|
+
_without_assignment(
|
|
491
|
+
replace(
|
|
492
|
+
state,
|
|
493
|
+
status="in_progress",
|
|
494
|
+
cursor=index,
|
|
495
|
+
active_item_id=plan.items[index].id,
|
|
496
|
+
item_executions=tuple(records),
|
|
497
|
+
last_error=None,
|
|
498
|
+
failure_kind=None,
|
|
499
|
+
updated_at=now(),
|
|
500
|
+
)
|
|
501
|
+
),
|
|
502
|
+
plan,
|
|
503
|
+
now,
|
|
504
|
+
)
|
|
505
|
+
|
|
506
|
+
|
|
507
|
+
def skip_idle_verification(state: ExecutionState, now: Clock) -> ExecutionState:
|
|
508
|
+
"""Pass a verification item that no round asked anything, such as after a
|
|
509
|
+
loop reset its record."""
|
|
510
|
+
records = list(state.item_executions)
|
|
511
|
+
records[state.cursor] = replace(
|
|
512
|
+
records[state.cursor],
|
|
513
|
+
status="completed",
|
|
514
|
+
completed_at=now(),
|
|
515
|
+
result=SKIPPED_ROUND,
|
|
516
|
+
)
|
|
517
|
+
return replace(
|
|
518
|
+
state,
|
|
519
|
+
cursor=state.cursor + 1,
|
|
520
|
+
item_executions=tuple(records),
|
|
521
|
+
updated_at=now(),
|
|
522
|
+
)
|
|
523
|
+
|
|
524
|
+
|
|
525
|
+
def _without_assignment(state: ExecutionState) -> ExecutionState:
|
|
526
|
+
return replace(
|
|
527
|
+
state,
|
|
528
|
+
assignment_item_id=None,
|
|
529
|
+
assignment_token=None,
|
|
530
|
+
assignment_model=None,
|
|
531
|
+
assignment_reasoning=None,
|
|
532
|
+
assignment_selected_agent=None,
|
|
533
|
+
assignment_selected_model=None,
|
|
534
|
+
assignment_selected_reasoning=None,
|
|
535
|
+
)
|
|
536
|
+
|
|
537
|
+
|
|
538
|
+
# --- Verifier results -------------------------------------------------------
|
|
539
|
+
|
|
540
|
+
|
|
541
|
+
@dataclass(frozen=True)
|
|
542
|
+
class JudgedFailure:
|
|
543
|
+
"""One piece of a verifier's evidence for a failing verdict."""
|
|
544
|
+
|
|
545
|
+
file: str
|
|
546
|
+
what: str
|
|
547
|
+
line: int | None = None
|
|
548
|
+
|
|
549
|
+
def text(self) -> str:
|
|
550
|
+
location = f"{self.file}:{self.line}" if self.line is not None else self.file
|
|
551
|
+
return f"{location} — {self.what}"
|
|
552
|
+
|
|
553
|
+
|
|
554
|
+
@dataclass(frozen=True)
|
|
555
|
+
class RuleResult:
|
|
556
|
+
"""A verifier's verdict on one rule, with its evidence when it fails."""
|
|
557
|
+
|
|
558
|
+
id: str
|
|
559
|
+
verdict: Verdict
|
|
560
|
+
failures: tuple[JudgedFailure, ...] = ()
|
|
561
|
+
|
|
562
|
+
|
|
563
|
+
_RULE_RESULT_KEYS = {"id", "status", "verdict", "failures"}
|
|
564
|
+
|
|
565
|
+
|
|
566
|
+
def parse_rule_results(
|
|
567
|
+
raw: tuple[str, ...], rules: tuple[VerificationRule, ...]
|
|
568
|
+
) -> tuple[RuleResult, ...]:
|
|
569
|
+
"""One validated result per rule the verification item covers, in order."""
|
|
570
|
+
by_id = {rule.id: rule for rule in rules}
|
|
571
|
+
parsed: dict[str, RuleResult] = {}
|
|
572
|
+
for text in raw:
|
|
573
|
+
data = _json_object(text, "--rule-result")
|
|
574
|
+
rule_id = data.get("id")
|
|
575
|
+
if not isinstance(rule_id, str) or not rule_id:
|
|
576
|
+
raise StateError("--rule-result requires the rule's id")
|
|
577
|
+
if rule_id not in by_id:
|
|
578
|
+
raise StateError(
|
|
579
|
+
f"--rule-result names {rule_id!r}, which this verification does "
|
|
580
|
+
"not cover; it covers " + ", ".join(by_id)
|
|
581
|
+
)
|
|
582
|
+
if rule_id in parsed:
|
|
583
|
+
raise StateError(f"--rule-result for {rule_id!r} is given twice")
|
|
584
|
+
parsed[rule_id] = _rule_result(data, by_id[rule_id])
|
|
585
|
+
missing = [rule.id for rule in rules if rule.id not in parsed]
|
|
586
|
+
if missing:
|
|
587
|
+
raise StateError("--rule-result is missing for " + ", ".join(missing))
|
|
588
|
+
return tuple(parsed[rule.id] for rule in rules)
|
|
589
|
+
|
|
590
|
+
|
|
591
|
+
def _rule_result(data: dict[str, Any], rule: VerificationRule) -> RuleResult:
|
|
592
|
+
label = f"--rule-result for {rule.id!r}"
|
|
593
|
+
# A verifier only judges: the status says so before anything else.
|
|
594
|
+
if data.get("status") != "judged":
|
|
595
|
+
raise StateError(f"{label}: status must be judged")
|
|
596
|
+
unknown = set(data) - _RULE_RESULT_KEYS
|
|
597
|
+
if unknown:
|
|
598
|
+
raise StateError(f"{label} has unknown keys: " + ", ".join(sorted(unknown)))
|
|
599
|
+
value = data.get("verdict")
|
|
600
|
+
if value not in {"pass", "fail"}:
|
|
601
|
+
raise StateError(f"{label}: judged requires verdict pass or fail")
|
|
602
|
+
verdict: Verdict = "pass" if value == "pass" else "fail"
|
|
603
|
+
return RuleResult(
|
|
604
|
+
id=rule.id,
|
|
605
|
+
verdict=verdict,
|
|
606
|
+
failures=_failures(data.get("failures"), label, verdict),
|
|
607
|
+
)
|
|
608
|
+
|
|
609
|
+
|
|
610
|
+
def _failures(value: Any, label: str, verdict: Verdict) -> tuple[JudgedFailure, ...]:
|
|
611
|
+
if verdict == "pass":
|
|
612
|
+
if value not in (None, []):
|
|
613
|
+
raise StateError(f"{label}: a pass verdict lists no failures")
|
|
614
|
+
return ()
|
|
615
|
+
if not isinstance(value, list) or not value:
|
|
616
|
+
raise StateError(
|
|
617
|
+
f"{label}: a fail verdict needs failures, each {{file, line?, what}}"
|
|
618
|
+
)
|
|
619
|
+
result = []
|
|
620
|
+
for entry in value:
|
|
621
|
+
if not isinstance(entry, dict) or not set(entry) <= {"file", "line", "what"}:
|
|
622
|
+
raise StateError(f"{label}: each failure is an object of file, line, what")
|
|
623
|
+
file, what, line = entry.get("file"), entry.get("what"), entry.get("line")
|
|
624
|
+
if not (isinstance(file, str) and file and isinstance(what, str) and what):
|
|
625
|
+
raise StateError(f"{label}: each failure needs a file and what")
|
|
626
|
+
if line is not None and (
|
|
627
|
+
not isinstance(line, int) or isinstance(line, bool) or line < 1
|
|
628
|
+
):
|
|
629
|
+
raise StateError(f"{label}: a failure's line is a positive integer")
|
|
630
|
+
result.append(JudgedFailure(file, what, line))
|
|
631
|
+
return tuple(result)
|
|
632
|
+
|
|
633
|
+
|
|
634
|
+
def _json_object(text: str, flag: str) -> dict[str, Any]:
|
|
635
|
+
try:
|
|
636
|
+
data = json.loads(text)
|
|
637
|
+
except json.JSONDecodeError as error:
|
|
638
|
+
raise StateError(f"{flag} is not valid JSON: {error}") from error
|
|
639
|
+
if not isinstance(data, dict):
|
|
640
|
+
raise StateError(f"{flag} must be a JSON object")
|
|
641
|
+
return data
|
|
642
|
+
|
|
643
|
+
|
|
644
|
+
def verdicts_of(results: tuple[RuleResult, ...], by: str) -> tuple[RuleVerdict, ...]:
|
|
645
|
+
return tuple(
|
|
646
|
+
RuleVerdict(
|
|
647
|
+
result.id,
|
|
648
|
+
result.verdict,
|
|
649
|
+
by,
|
|
650
|
+
tuple(failure.text() for failure in result.failures),
|
|
651
|
+
)
|
|
652
|
+
for result in results
|
|
653
|
+
)
|
|
654
|
+
|
|
655
|
+
|
|
656
|
+
def judged_report(
|
|
657
|
+
verdicts: tuple[RuleVerdict, ...],
|
|
658
|
+
attempt: int,
|
|
659
|
+
checked_at: str,
|
|
660
|
+
held: HeldCompletion,
|
|
661
|
+
) -> CheckReport:
|
|
662
|
+
"""A round's verdicts as a check report, so failures join the fix loop."""
|
|
663
|
+
return CheckReport(
|
|
664
|
+
attempt=attempt,
|
|
665
|
+
checked_at=checked_at,
|
|
666
|
+
results=tuple(
|
|
667
|
+
CheckResult(
|
|
668
|
+
verdict.id,
|
|
669
|
+
"judged",
|
|
670
|
+
"passed" if verdict.verdict == "pass" else "failed",
|
|
671
|
+
output="\n".join(verdict.failures),
|
|
672
|
+
)
|
|
673
|
+
for verdict in verdicts
|
|
674
|
+
),
|
|
675
|
+
mark=held.mark,
|
|
676
|
+
all_files=held.all_files,
|
|
677
|
+
)
|
|
678
|
+
|
|
679
|
+
|
|
680
|
+
# --- Revoking a check ----------------------------------------------------------
|
|
681
|
+
|
|
682
|
+
|
|
683
|
+
def revoke_check(
|
|
684
|
+
automation: RuleAutomation, name: str, reason: str
|
|
685
|
+
) -> tuple[RuleAutomation, tuple[str, ...]]:
|
|
686
|
+
"""Reject a converted or proposed check and the rules it covers.
|
|
687
|
+
|
|
688
|
+
The rules are judged by a verifier from then on. A pending revision is
|
|
689
|
+
dropped with it. Returns the store and the text hashes of the rules
|
|
690
|
+
rejected. Nothing outside the store changes.
|
|
691
|
+
"""
|
|
692
|
+
check = automation.checks.get(name)
|
|
693
|
+
if check is None:
|
|
694
|
+
raise StateError(f"the rule-automation store has no check {name!r}")
|
|
695
|
+
if check.status == "rejected":
|
|
696
|
+
raise StateError(f"check {name!r} is already rejected")
|
|
697
|
+
covers = tuple(
|
|
698
|
+
dict.fromkeys(
|
|
699
|
+
(*check.spec.covers, *(check.pending.covers if check.pending else ()))
|
|
700
|
+
)
|
|
701
|
+
)
|
|
702
|
+
automation = automation.with_check(
|
|
703
|
+
name, replace(check, status="rejected", pending=None, reason=reason)
|
|
704
|
+
)
|
|
705
|
+
rejected: list[str] = []
|
|
706
|
+
for text_hash in covers:
|
|
707
|
+
entry = automation.rules.get(text_hash)
|
|
708
|
+
if entry is None or entry.check != name or entry.status == "rejected":
|
|
709
|
+
continue
|
|
710
|
+
automation = automation.with_rule(
|
|
711
|
+
text_hash, replace(entry, status="rejected", reason=reason)
|
|
712
|
+
)
|
|
713
|
+
rejected.append(text_hash)
|
|
714
|
+
return automation, tuple(rejected)
|