provledger 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- provledger-0.1.0/PKG-INFO +14 -0
- provledger-0.1.0/orchestrator/__init__.py +5 -0
- provledger-0.1.0/orchestrator/api.py +661 -0
- provledger-0.1.0/orchestrator/circuit_breakers.py +56 -0
- provledger-0.1.0/orchestrator/data_loop.py +81 -0
- provledger-0.1.0/orchestrator/db.py +543 -0
- provledger-0.1.0/orchestrator/drift.py +71 -0
- provledger-0.1.0/orchestrator/migrations/001_init_schema.sql +35 -0
- provledger-0.1.0/orchestrator/migrations/002_step_types_and_agent_io.sql +19 -0
- provledger-0.1.0/orchestrator/migrations/003_user_query.sql +13 -0
- provledger-0.1.0/orchestrator/migrations/004_step_summary.sql +22 -0
- provledger-0.1.0/orchestrator/migrations/005_skill_activations.sql +45 -0
- provledger-0.1.0/orchestrator/migrations/006_review_step.sql +30 -0
- provledger-0.1.0/orchestrator/migrations/007_needs_review_status.sql +70 -0
- provledger-0.1.0/orchestrator/migrations/008_impact_context.sql +13 -0
- provledger-0.1.0/orchestrator/migrations/009_ledger_entries.sql +35 -0
- provledger-0.1.0/orchestrator/migrations/010_data_model.sql +60 -0
- provledger-0.1.0/orchestrator/migrations/011_completed_immutability.sql +14 -0
- provledger-0.1.0/orchestrator/migrations/012_data_profile.sql +24 -0
- provledger-0.1.0/orchestrator/migrations/013_llm_decisions.sql +28 -0
- provledger-0.1.0/orchestrator/profiler.py +89 -0
- provledger-0.1.0/orchestrator/state_machine.py +54 -0
- provledger-0.1.0/orchestrator/telemetry.py +58 -0
- provledger-0.1.0/provledger.egg-info/PKG-INFO +14 -0
- provledger-0.1.0/provledger.egg-info/SOURCES.txt +47 -0
- provledger-0.1.0/provledger.egg-info/dependency_links.txt +1 -0
- provledger-0.1.0/provledger.egg-info/top_level.txt +1 -0
- provledger-0.1.0/pyproject.toml +45 -0
- provledger-0.1.0/setup.cfg +4 -0
- provledger-0.1.0/tests/test_api.py +222 -0
- provledger-0.1.0/tests/test_circuit_breakers.py +60 -0
- provledger-0.1.0/tests/test_data_loop.py +102 -0
- provledger-0.1.0/tests/test_data_model_wiring.py +42 -0
- provledger-0.1.0/tests/test_db.py +108 -0
- provledger-0.1.0/tests/test_detect_registered_project.py +108 -0
- provledger-0.1.0/tests/test_drift.py +48 -0
- provledger-0.1.0/tests/test_llm_decisions.py +57 -0
- provledger-0.1.0/tests/test_migration_006_review_step.py +118 -0
- provledger-0.1.0/tests/test_migration_008_impact_context.py +40 -0
- provledger-0.1.0/tests/test_migration_009_ledger.py +40 -0
- provledger-0.1.0/tests/test_migration_010.py +73 -0
- provledger-0.1.0/tests/test_migration_011.py +40 -0
- provledger-0.1.0/tests/test_profiler.py +53 -0
- provledger-0.1.0/tests/test_review_child_step.py +152 -0
- provledger-0.1.0/tests/test_review_path_selection.py +102 -0
- provledger-0.1.0/tests/test_review_recovery.py +205 -0
- provledger-0.1.0/tests/test_skill_activations.py +178 -0
- provledger-0.1.0/tests/test_state_machine.py +96 -0
- provledger-0.1.0/tests/test_telemetry.py +92 -0
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: provledger
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Data-contract + provenance layer for data-science coding agents: SQLite plan/step orchestrator, runtime data profiling, drift detection, and a never-repeat-a-mistake decision ledger
|
|
5
|
+
Author-email: yizhao95 <yzhao950213@gmail.com>
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/yizhao95/prov_ledger
|
|
8
|
+
Project-URL: Documentation, https://github.com/yizhao95/prov_ledger#readme
|
|
9
|
+
Keywords: data-contracts,provenance,llm-agents,data-quality,drift-detection
|
|
10
|
+
Classifier: Development Status :: 3 - Alpha
|
|
11
|
+
Classifier: Intended Audience :: Developers
|
|
12
|
+
Classifier: Programming Language :: Python :: 3
|
|
13
|
+
Classifier: Topic :: Scientific/Engineering
|
|
14
|
+
Requires-Python: >=3.10
|
|
@@ -0,0 +1,661 @@
|
|
|
1
|
+
"""High-level harness API — `initialize_plan` + `evaluate_and_update_plan`.
|
|
2
|
+
|
|
3
|
+
Implements the two strictly-typed contracts from Phase 1.docx, plus convenience
|
|
4
|
+
helpers that wrap state_machine + circuit_breakers + telemetry.
|
|
5
|
+
"""
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
import json
|
|
9
|
+
import os
|
|
10
|
+
import re
|
|
11
|
+
import sqlite3
|
|
12
|
+
import string
|
|
13
|
+
from datetime import datetime, timezone
|
|
14
|
+
|
|
15
|
+
from . import circuit_breakers, db, state_machine, telemetry
|
|
16
|
+
from .circuit_breakers import HardStop, SoftStop # noqa: F401 re-export
|
|
17
|
+
from .state_machine import InvalidTransitionError, StepStatus # noqa: F401
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _now_compact() -> str:
|
|
21
|
+
return datetime.now(timezone.utc).strftime("%Y%m%d%H%M%S")
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def _step_label(index: int) -> str:
|
|
25
|
+
"""0→'A', 1→'B', ..., 25→'Z', 26→'AA', etc."""
|
|
26
|
+
if index < 26:
|
|
27
|
+
return string.ascii_uppercase[index]
|
|
28
|
+
first, second = divmod(index, 26)
|
|
29
|
+
return string.ascii_uppercase[first - 1] + string.ascii_uppercase[second]
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
# ── Tool 1: initialize_plan ────────────────────────────────────────────
|
|
33
|
+
def _step_spec(spec) -> tuple[str, str | None]:
|
|
34
|
+
"""Normalize a step spec — a bare description string, or a dict of
|
|
35
|
+
{description, step_type?} — to (description, step_type).
|
|
36
|
+
step_type validity is enforced by db.insert_step."""
|
|
37
|
+
if isinstance(spec, str):
|
|
38
|
+
return spec, None
|
|
39
|
+
return spec["description"], spec.get("step_type")
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def initialize_plan(
|
|
43
|
+
conn: sqlite3.Connection,
|
|
44
|
+
original_goal: str,
|
|
45
|
+
initial_steps: list,
|
|
46
|
+
plan_id_prefix: str = "plan",
|
|
47
|
+
max_revisions: int = 5,
|
|
48
|
+
user_query: str | None = None,
|
|
49
|
+
skills_activated: list[dict] | None = None,
|
|
50
|
+
) -> dict:
|
|
51
|
+
"""Create a new plan with N top-level steps. Returns {plan_id, step_ids}.
|
|
52
|
+
|
|
53
|
+
`original_goal` is the agent's one-line summary of intent.
|
|
54
|
+
`initial_steps` items are bare description strings, or dicts of
|
|
55
|
+
{description, step_type?} so a type declared at plan time lands at publish
|
|
56
|
+
(previously it was silently dropped; run-step may still set/override the
|
|
57
|
+
type at execution time).
|
|
58
|
+
`user_query` is the verbatim prompt the human typed (optional, used as
|
|
59
|
+
history-page title).
|
|
60
|
+
`skills_activated` is an optional list of dicts recording which skills were
|
|
61
|
+
already loaded BEFORE the plan was created (e.g., writing-plans + TDD via
|
|
62
|
+
iron-law). Each dict requires keys: skill_name, source. Optional: reason.
|
|
63
|
+
All entries are recorded with step_id=NULL (init-time activation).
|
|
64
|
+
"""
|
|
65
|
+
if not initial_steps:
|
|
66
|
+
raise ValueError("initial_steps must contain at least 1 step")
|
|
67
|
+
plan_id = f"{plan_id_prefix}-{_now_compact()}"
|
|
68
|
+
db.insert_plan(conn, plan_id, original_goal, max_revisions=max_revisions, user_query=user_query)
|
|
69
|
+
step_ids = []
|
|
70
|
+
for i, spec in enumerate(initial_steps):
|
|
71
|
+
desc, step_type = _step_spec(spec)
|
|
72
|
+
sid = f"{plan_id}-{_step_label(i)}"
|
|
73
|
+
db.insert_step(
|
|
74
|
+
conn, sid, plan_id, desc,
|
|
75
|
+
execution_order=i, depth_level=0, parent_step_id=None,
|
|
76
|
+
step_type=step_type,
|
|
77
|
+
)
|
|
78
|
+
step_ids.append(sid)
|
|
79
|
+
|
|
80
|
+
# Record any pre-activated skills (init-time, step_id=NULL)
|
|
81
|
+
if skills_activated:
|
|
82
|
+
for entry in skills_activated:
|
|
83
|
+
db.add_skill_activation(
|
|
84
|
+
conn,
|
|
85
|
+
plan_id=plan_id,
|
|
86
|
+
skill_name=entry["skill_name"],
|
|
87
|
+
source=entry["source"],
|
|
88
|
+
step_id=None,
|
|
89
|
+
reason=entry.get("reason"),
|
|
90
|
+
)
|
|
91
|
+
|
|
92
|
+
return {"plan_id": plan_id, "step_ids": step_ids}
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
# ── Skill activation helper ──────────────────────────────────────────────
|
|
96
|
+
def record_skill_activation(
|
|
97
|
+
conn: sqlite3.Connection,
|
|
98
|
+
plan_id: str,
|
|
99
|
+
skill_name: str,
|
|
100
|
+
source: str,
|
|
101
|
+
step_id: str | None = None,
|
|
102
|
+
reason: str | None = None,
|
|
103
|
+
) -> int:
|
|
104
|
+
"""Record one mid-execution skill activation. Thin wrapper over db.add_skill_activation.
|
|
105
|
+
|
|
106
|
+
Use during execution when a step triggers a new skill that wasn't pre-loaded
|
|
107
|
+
at plan-init time (e.g., systematic-debugging triggered by an error in step F,
|
|
108
|
+
or a sub-agent loading a domain skill). For init-time activations, prefer
|
|
109
|
+
passing them via `initialize_plan(skills_activated=[...])`.
|
|
110
|
+
|
|
111
|
+
Returns the new activation_id.
|
|
112
|
+
"""
|
|
113
|
+
return db.add_skill_activation(
|
|
114
|
+
conn,
|
|
115
|
+
plan_id=plan_id,
|
|
116
|
+
skill_name=skill_name,
|
|
117
|
+
source=source,
|
|
118
|
+
step_id=step_id,
|
|
119
|
+
reason=reason,
|
|
120
|
+
)
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
# ── Tool 2: evaluate_and_update_plan ──────────────────────────────────────────
|
|
124
|
+
def evaluate_and_update_plan(
|
|
125
|
+
conn: sqlite3.Connection,
|
|
126
|
+
deviation_detected: bool,
|
|
127
|
+
target_step_id: str | None = None,
|
|
128
|
+
justification: str | None = None,
|
|
129
|
+
new_sub_steps: list | None = None,
|
|
130
|
+
) -> dict:
|
|
131
|
+
"""Register a deviation; optionally insert sub-steps. Enforces all 3 circuit
|
|
132
|
+
breakers. Sub-step items are description strings or {description, step_type?}
|
|
133
|
+
dicts (same shapes as initialize_plan's initial_steps)."""
|
|
134
|
+
if not deviation_detected:
|
|
135
|
+
return {"accepted": True, "no_changes": True}
|
|
136
|
+
if not target_step_id or not justification:
|
|
137
|
+
raise ValueError("deviation_detected requires both target_step_id AND justification")
|
|
138
|
+
|
|
139
|
+
target = db.get_step(conn, target_step_id)
|
|
140
|
+
if not target:
|
|
141
|
+
return {"accepted": False, "reason": f"step_id not found: {target_step_id}"}
|
|
142
|
+
plan = db.get_plan(conn, target["plan_id"])
|
|
143
|
+
if not plan:
|
|
144
|
+
return {"accepted": False, "reason": f"plan_id not found: {target['plan_id']}"}
|
|
145
|
+
|
|
146
|
+
# ── Circuit breakers (raise on violation) ────────────────────────────────
|
|
147
|
+
try:
|
|
148
|
+
circuit_breakers.check_immutability(target["status"])
|
|
149
|
+
warning = circuit_breakers.check_loop_prevention(
|
|
150
|
+
plan["revision_count"], plan["max_revisions"]
|
|
151
|
+
)
|
|
152
|
+
if new_sub_steps:
|
|
153
|
+
circuit_breakers.check_depth_limit(target["depth_level"])
|
|
154
|
+
except SoftStop as e:
|
|
155
|
+
return {"accepted": False, "reason": str(e), "breaker": "soft"}
|
|
156
|
+
# HardStop is intentionally NOT caught — it propagates up as agent-pause signal
|
|
157
|
+
|
|
158
|
+
new_step_ids: list[str] = []
|
|
159
|
+
existing_children = db.get_children(conn, target_step_id) if new_sub_steps else []
|
|
160
|
+
# BE-C4: sub-step inserts + revision bump + deviation record are one atomic
|
|
161
|
+
# unit — an interruption mid-sequence must not leave the plan half-mutated.
|
|
162
|
+
with db.transaction(conn):
|
|
163
|
+
for i, spec in enumerate(new_sub_steps or []):
|
|
164
|
+
desc, step_type = _step_spec(spec)
|
|
165
|
+
sid = f"{target_step_id}.{len(existing_children) + i + 1}"
|
|
166
|
+
db.insert_step(
|
|
167
|
+
conn, sid, target["plan_id"], desc,
|
|
168
|
+
execution_order=target["execution_order"] * 100 + i,
|
|
169
|
+
parent_step_id=target_step_id,
|
|
170
|
+
depth_level=target["depth_level"] + 1,
|
|
171
|
+
step_type=step_type,
|
|
172
|
+
commit=False,
|
|
173
|
+
)
|
|
174
|
+
new_step_ids.append(sid)
|
|
175
|
+
new_revision = db.increment_revision(conn, plan["plan_id"], commit=False)
|
|
176
|
+
deviation_id = db.insert_deviation(
|
|
177
|
+
conn, plan["plan_id"], target_step_id, justification,
|
|
178
|
+
new_step_ids=new_step_ids, revision_count=new_revision, commit=False,
|
|
179
|
+
)
|
|
180
|
+
result: dict = {
|
|
181
|
+
"accepted": True,
|
|
182
|
+
"new_step_ids": new_step_ids,
|
|
183
|
+
"justification_logged": justification,
|
|
184
|
+
"deviation_id": deviation_id,
|
|
185
|
+
"revision_count": new_revision,
|
|
186
|
+
}
|
|
187
|
+
if warning:
|
|
188
|
+
result["warning"] = warning
|
|
189
|
+
return result
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
# ── Convenience wrappers (validated transitions) ──────────────────────────────
|
|
193
|
+
def start_step(conn: sqlite3.Connection, step_id: str) -> dict:
|
|
194
|
+
"""PENDING → IN_PROGRESS, sets started_at."""
|
|
195
|
+
step = db.get_step(conn, step_id)
|
|
196
|
+
if not step:
|
|
197
|
+
raise ValueError(f"step_id not found: {step_id}")
|
|
198
|
+
state_machine.validate_transition(step["status"], "IN_PROGRESS")
|
|
199
|
+
db.update_step_status(conn, step_id, "IN_PROGRESS", set_started=True)
|
|
200
|
+
return db.get_step(conn, step_id)
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
def complete_step(conn: sqlite3.Connection, step_id: str) -> dict:
|
|
204
|
+
"""IN_PROGRESS → COMPLETED, sets completed_at."""
|
|
205
|
+
step = db.get_step(conn, step_id)
|
|
206
|
+
if not step:
|
|
207
|
+
raise ValueError(f"step_id not found: {step_id}")
|
|
208
|
+
circuit_breakers.check_immutability(step["status"]) # no double-completing
|
|
209
|
+
state_machine.validate_transition(step["status"], "COMPLETED")
|
|
210
|
+
db.update_step_status(conn, step_id, "COMPLETED", set_completed=True)
|
|
211
|
+
return db.get_step(conn, step_id)
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
def fail_step(conn: sqlite3.Connection, step_id: str, reason: str = "") -> dict:
|
|
215
|
+
"""Fail a started step (STARTING/IN_PROGRESS/NEEDS_REVIEW → FAILED).
|
|
216
|
+
|
|
217
|
+
PENDING steps cannot be failed — the state machine rejects PENDING → FAILED
|
|
218
|
+
(start the step first). Persists the reason to Steps.failure_reason and the log.
|
|
219
|
+
"""
|
|
220
|
+
step = db.get_step(conn, step_id)
|
|
221
|
+
if not step:
|
|
222
|
+
raise ValueError(f"step_id not found: {step_id}")
|
|
223
|
+
state_machine.validate_transition(step["status"], "FAILED")
|
|
224
|
+
db.update_step_status(conn, step_id, "FAILED", set_completed=True)
|
|
225
|
+
if reason:
|
|
226
|
+
db.set_failure_reason(conn, step_id, reason)
|
|
227
|
+
telemetry.append_step_log(conn, step_id, f"[FAILED] {reason}")
|
|
228
|
+
return db.get_step(conn, step_id)
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
def append_log(conn: sqlite3.Connection, step_id: str, raw_chunk: str) -> str:
|
|
232
|
+
"""Append telemetry to step's log_context (with truncation)."""
|
|
233
|
+
return telemetry.append_step_log(conn, step_id, raw_chunk)
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
def complete_plan(conn: sqlite3.Connection, plan_id: str) -> dict:
|
|
237
|
+
"""Mark plan COMPLETED. Doesn't validate that all steps are done — caller's job."""
|
|
238
|
+
db.update_plan_status(conn, plan_id, "COMPLETED")
|
|
239
|
+
return db.get_plan(conn, plan_id)
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
# ── Deterministic auto-review-and-complete procedure (migration 006) ───────────────
|
|
243
|
+
TERMINAL_STEP_STATES = {"COMPLETED", "FAILED"}
|
|
244
|
+
|
|
245
|
+
|
|
246
|
+
def _is_step_recovered(conn: sqlite3.Connection, step_id: str) -> bool:
|
|
247
|
+
"""True if the step's outcome should be treated as success for plan-level rollup.
|
|
248
|
+
|
|
249
|
+
Rules:
|
|
250
|
+
- COMPLETED → recovered (trivially)
|
|
251
|
+
- FAILED with zero non-review children → NOT recovered (failure stuck)
|
|
252
|
+
- FAILED with all non-review children recovered (recursive) → recovered
|
|
253
|
+
- FAILED with at least one unrecovered/non-terminal child → NOT recovered
|
|
254
|
+
- Any non-terminal state (PENDING / IN_PROGRESS / STARTING) → NOT recovered
|
|
255
|
+
|
|
256
|
+
Recursion is naturally bounded by the depth_level<=3 circuit breaker in
|
|
257
|
+
executing-plans, so this function will not pathologically deep-recurse.
|
|
258
|
+
"""
|
|
259
|
+
row = conn.execute(
|
|
260
|
+
"SELECT status FROM Steps WHERE step_id = ?", (step_id,)
|
|
261
|
+
).fetchone()
|
|
262
|
+
if row is None:
|
|
263
|
+
return False
|
|
264
|
+
status = row["status"]
|
|
265
|
+
if status == "COMPLETED":
|
|
266
|
+
return True
|
|
267
|
+
if status != "FAILED":
|
|
268
|
+
# PENDING / IN_PROGRESS / STARTING — not terminal, cannot be 'recovered'
|
|
269
|
+
return False
|
|
270
|
+
# FAILED: check deviation sub-tree
|
|
271
|
+
children = conn.execute(
|
|
272
|
+
"SELECT step_id FROM Steps WHERE parent_step_id = ? AND is_review = 0",
|
|
273
|
+
(step_id,),
|
|
274
|
+
).fetchall()
|
|
275
|
+
if not children:
|
|
276
|
+
return False # FAILED leaf — no deviation, not recovered
|
|
277
|
+
return all(_is_step_recovered(conn, c["step_id"]) for c in children)
|
|
278
|
+
|
|
279
|
+
|
|
280
|
+
DEFAULT_REGISTRY_PATH = os.path.expanduser(
|
|
281
|
+
"~/skill-workspace/project-graphs/projects.json"
|
|
282
|
+
)
|
|
283
|
+
|
|
284
|
+
# Sentinel so callers can pass registry_path=None to mean "use the default";
|
|
285
|
+
# the default itself honors the PSG_REGISTRY_PATH env override (test isolation +
|
|
286
|
+
# parity with project-state-graph's env-overridable registry paths).
|
|
287
|
+
_REGISTRY_DEFAULT = object()
|
|
288
|
+
|
|
289
|
+
|
|
290
|
+
def _resolve_registry_path(registry_path) -> str:
|
|
291
|
+
if registry_path is _REGISTRY_DEFAULT or registry_path is None:
|
|
292
|
+
return os.environ.get("PSG_REGISTRY_PATH", DEFAULT_REGISTRY_PATH)
|
|
293
|
+
return registry_path
|
|
294
|
+
|
|
295
|
+
|
|
296
|
+
def _normalize_project_token(s: str) -> str:
|
|
297
|
+
"""Canonicalize a project name / text for variance-tolerant matching.
|
|
298
|
+
|
|
299
|
+
Lowercase and strip all hyphens, underscores, and whitespace so that
|
|
300
|
+
'demo-app', 'demo app', 'demo_app', 'DEMOAPP' all collapse to
|
|
301
|
+
the same token 'demoapp'.
|
|
302
|
+
"""
|
|
303
|
+
return re.sub(r"[-_\s]+", "", s.lower())
|
|
304
|
+
|
|
305
|
+
|
|
306
|
+
def detect_registered_project(
|
|
307
|
+
conn: sqlite3.Connection,
|
|
308
|
+
plan_id: str,
|
|
309
|
+
registry_path=_REGISTRY_DEFAULT,
|
|
310
|
+
) -> str | None:
|
|
311
|
+
"""Return the canonical name of a registered project mentioned by the plan.
|
|
312
|
+
|
|
313
|
+
Scans the plan goal + every NON-review step description for a mention of any
|
|
314
|
+
project in the registry (projects.json), using variance-tolerant matching
|
|
315
|
+
(see _normalize_project_token). Returns the first registered project's
|
|
316
|
+
canonical name found, or None if no registered project is mentioned (or the
|
|
317
|
+
registry is empty/absent).
|
|
318
|
+
|
|
319
|
+
Deterministic — no LLM. Used by review_and_complete to decide whether plan
|
|
320
|
+
completion needs an LLM sub-agent review.
|
|
321
|
+
"""
|
|
322
|
+
registry_path = _resolve_registry_path(registry_path)
|
|
323
|
+
if not registry_path or not os.path.exists(registry_path):
|
|
324
|
+
return None
|
|
325
|
+
try:
|
|
326
|
+
with open(registry_path) as f:
|
|
327
|
+
registry = json.load(f)
|
|
328
|
+
except (json.JSONDecodeError, OSError):
|
|
329
|
+
return None
|
|
330
|
+
projects = registry.get("projects", []) if isinstance(registry, dict) else []
|
|
331
|
+
if not projects:
|
|
332
|
+
return None
|
|
333
|
+
|
|
334
|
+
# Build {normalized_name: canonical_name}, preserving registry order.
|
|
335
|
+
normalized = []
|
|
336
|
+
for p in projects:
|
|
337
|
+
name = p.get("name")
|
|
338
|
+
if name:
|
|
339
|
+
normalized.append((_normalize_project_token(name), name))
|
|
340
|
+
if not normalized:
|
|
341
|
+
return None
|
|
342
|
+
|
|
343
|
+
# Gather plan text: goal + non-review step descriptions.
|
|
344
|
+
plan = db.get_plan(conn, plan_id)
|
|
345
|
+
texts = []
|
|
346
|
+
if plan and plan.get("original_goal"):
|
|
347
|
+
texts.append(plan["original_goal"])
|
|
348
|
+
step_rows = conn.execute(
|
|
349
|
+
"SELECT description FROM Steps "
|
|
350
|
+
"WHERE plan_id = ? AND is_review = 0 AND description IS NOT NULL",
|
|
351
|
+
(plan_id,),
|
|
352
|
+
).fetchall()
|
|
353
|
+
texts.extend(r["description"] for r in step_rows)
|
|
354
|
+
|
|
355
|
+
# BE-S1: match on word boundaries, not raw substring. Splitting into word
|
|
356
|
+
# tokens (then matching a project's collapsed name against a CONTIGUOUS RUN of
|
|
357
|
+
# tokens) keeps variance tolerance — "demo app" / "demo-app" / "demoapp" all
|
|
358
|
+
# match "demo-app" — while preventing a short name like "app" from matching
|
|
359
|
+
# inside an unrelated word like "happens".
|
|
360
|
+
hay_words = [w for w in re.split(r"[-_\s]+", " ".join(texts).lower()) if w]
|
|
361
|
+
for norm_name, canonical in normalized:
|
|
362
|
+
if norm_name and _matches_token_run(hay_words, norm_name):
|
|
363
|
+
return canonical
|
|
364
|
+
return None
|
|
365
|
+
|
|
366
|
+
|
|
367
|
+
def _matches_token_run(words: list[str], target: str) -> bool:
|
|
368
|
+
"""True if `target` equals the concatenation of some contiguous run of `words`.
|
|
369
|
+
|
|
370
|
+
`target` is an already-collapsed project name (no separators). A single-word
|
|
371
|
+
project matches a standalone token; a multi-word project matches adjacent
|
|
372
|
+
tokens whose concatenation equals it.
|
|
373
|
+
"""
|
|
374
|
+
n = len(words)
|
|
375
|
+
for i in range(n):
|
|
376
|
+
acc = ""
|
|
377
|
+
for j in range(i, n):
|
|
378
|
+
acc += words[j]
|
|
379
|
+
if len(acc) > len(target):
|
|
380
|
+
break
|
|
381
|
+
if acc == target:
|
|
382
|
+
return True
|
|
383
|
+
return False
|
|
384
|
+
|
|
385
|
+
|
|
386
|
+
def _open_agent_review(conn: sqlite3.Connection, plan_id: str, review_step_id: str) -> str:
|
|
387
|
+
"""Enter the tracked agent-review phase for a registered-project plan.
|
|
388
|
+
|
|
389
|
+
Deterministically:
|
|
390
|
+
- flip the review step to NEEDS_REVIEW,
|
|
391
|
+
- leave the plan IN_PROGRESS,
|
|
392
|
+
- bump the plan revision (this IS a plan revision — we add a step),
|
|
393
|
+
- insert a child SUB_AGENT step <review>.1 (PENDING, depth 1) that the
|
|
394
|
+
review sub-agent will drive.
|
|
395
|
+
|
|
396
|
+
Idempotent: if a child step already exists under the review step, do NOT
|
|
397
|
+
re-insert it and do NOT bump the revision again. Returns the child step_id.
|
|
398
|
+
"""
|
|
399
|
+
existing = db.get_children(conn, review_step_id)
|
|
400
|
+
if existing:
|
|
401
|
+
# Already opened — return the first child (there is only ever one).
|
|
402
|
+
return existing[0]["step_id"]
|
|
403
|
+
|
|
404
|
+
review_row = db.get_step(conn, review_step_id)
|
|
405
|
+
child_id = f"{review_step_id}.1"
|
|
406
|
+
# BE-C4: the review-step flip + plan-state writes + child insert are one
|
|
407
|
+
# atomic unit; a crash mid-sequence must not park a plan with no child step.
|
|
408
|
+
with db.transaction(conn):
|
|
409
|
+
db.update_step_status(conn, review_step_id, "NEEDS_REVIEW", commit=False)
|
|
410
|
+
db.update_plan_status(conn, plan_id, "IN_PROGRESS", commit=False)
|
|
411
|
+
db.set_review_state(conn, plan_id, "awaiting_agent", commit=False) # BE-D4
|
|
412
|
+
db.increment_revision(conn, plan_id, commit=False)
|
|
413
|
+
db.insert_step(
|
|
414
|
+
conn,
|
|
415
|
+
child_id,
|
|
416
|
+
plan_id,
|
|
417
|
+
"AGENT REVIEW: project-state-graph consistency review (LLM sub-agent)",
|
|
418
|
+
execution_order=(review_row["execution_order"] or 0) * 100,
|
|
419
|
+
parent_step_id=review_step_id,
|
|
420
|
+
depth_level=(review_row["depth_level"] or 0) + 1,
|
|
421
|
+
step_type="SUB_AGENT",
|
|
422
|
+
commit=False,
|
|
423
|
+
)
|
|
424
|
+
return child_id
|
|
425
|
+
|
|
426
|
+
|
|
427
|
+
def review_and_complete(
|
|
428
|
+
conn: sqlite3.Connection,
|
|
429
|
+
plan_id: str,
|
|
430
|
+
registry_path=_REGISTRY_DEFAULT,
|
|
431
|
+
) -> dict:
|
|
432
|
+
"""Deterministic 'review and complete' procedure.
|
|
433
|
+
|
|
434
|
+
Examines every non-review step in plan_id and decides the plan's terminal
|
|
435
|
+
state purely from data:
|
|
436
|
+
|
|
437
|
+
- all non-review steps COMPLETED → review COMPLETED, plan COMPLETED
|
|
438
|
+
- any non-review step FAILED but the failure is RECOVERED via a deviation
|
|
439
|
+
sub-tree (see _is_step_recovered) → review COMPLETED, plan COMPLETED
|
|
440
|
+
- any non-review step FAILED & UNRECOVERED (no children, or some child
|
|
441
|
+
also failed without its own recovery) → review FAILED, plan FAILED
|
|
442
|
+
(provided all other non-review steps are in some terminal state)
|
|
443
|
+
- any non-review step still PENDING
|
|
444
|
+
or IN_PROGRESS → no-op, returns ready=False
|
|
445
|
+
|
|
446
|
+
LLM-review routing (project-state-graph): when the plan WOULD close as
|
|
447
|
+
COMPLETED *and* it mentions a registered project (detect_registered_project),
|
|
448
|
+
the review step is instead flipped to NEEDS_REVIEW and the plan is LEFT
|
|
449
|
+
IN_PROGRESS. The result carries needs_agent_review=True + project so the
|
|
450
|
+
executing-plans main agent knows to dispatch the review sub-agent, which
|
|
451
|
+
finalizes the plan via agent-review-close.sh. The FAILED path is never
|
|
452
|
+
intercepted — unrecovered failures always propagate immediately.
|
|
453
|
+
|
|
454
|
+
Idempotent: re-calling when the review step is already terminal OR already
|
|
455
|
+
NEEDS_REVIEW returns the current state without re-mutating.
|
|
456
|
+
"""
|
|
457
|
+
# Find the review step (at most one per plan; enforced by step_id uniqueness)
|
|
458
|
+
review_row = conn.execute(
|
|
459
|
+
"SELECT step_id, status FROM Steps WHERE plan_id = ? AND is_review = 1 LIMIT 1",
|
|
460
|
+
(plan_id,),
|
|
461
|
+
).fetchone()
|
|
462
|
+
if review_row is None:
|
|
463
|
+
return {
|
|
464
|
+
"ready": False,
|
|
465
|
+
"plan_status": (db.get_plan(conn, plan_id) or {}).get("status"),
|
|
466
|
+
"review_step_id": None,
|
|
467
|
+
"review_status": None,
|
|
468
|
+
"reason": "no review step (pre-migration-006 plan?)",
|
|
469
|
+
}
|
|
470
|
+
|
|
471
|
+
review_step_id = review_row["step_id"]
|
|
472
|
+
review_status = review_row["status"]
|
|
473
|
+
|
|
474
|
+
# Idempotency: if review step already finalized, return current state
|
|
475
|
+
if review_status in TERMINAL_STEP_STATES:
|
|
476
|
+
plan_row = db.get_plan(conn, plan_id)
|
|
477
|
+
return {
|
|
478
|
+
"ready": True,
|
|
479
|
+
"plan_status": plan_row["status"] if plan_row else None,
|
|
480
|
+
"review_step_id": review_step_id,
|
|
481
|
+
"review_status": review_status,
|
|
482
|
+
"reason": "review step already terminal (idempotent no-op)",
|
|
483
|
+
}
|
|
484
|
+
|
|
485
|
+
# Agent-review phase: if the review step is awaiting an LLM sub-agent, the
|
|
486
|
+
# plan's terminal state is driven by the tracked child step <review>.1.
|
|
487
|
+
# This MUST run before the sibling tally below, because the child step has
|
|
488
|
+
# is_review=0 and would otherwise be miscounted as a regular sibling.
|
|
489
|
+
if review_status == "NEEDS_REVIEW":
|
|
490
|
+
children = db.get_children(conn, review_step_id)
|
|
491
|
+
child = children[0] if children else None
|
|
492
|
+
if child is None:
|
|
493
|
+
# Heal-forward: a plan parked in NEEDS_REVIEW by older code with no
|
|
494
|
+
# child step. Open the tracked review now (idempotent).
|
|
495
|
+
child_id = _open_agent_review(conn, plan_id, review_step_id)
|
|
496
|
+
return {
|
|
497
|
+
"ready": False,
|
|
498
|
+
"needs_agent_review": True,
|
|
499
|
+
"project": detect_registered_project(conn, plan_id, registry_path),
|
|
500
|
+
"plan_status": "IN_PROGRESS",
|
|
501
|
+
"review_step_id": review_step_id,
|
|
502
|
+
"review_child_step_id": child_id,
|
|
503
|
+
"review_status": "NEEDS_REVIEW",
|
|
504
|
+
"reason": "review awaiting agent; child review step (re)created",
|
|
505
|
+
}
|
|
506
|
+
if child["status"] == "COMPLETED":
|
|
507
|
+
db.update_step_status(conn, review_step_id, "COMPLETED", set_completed=True)
|
|
508
|
+
db.update_plan_status(conn, plan_id, "COMPLETED")
|
|
509
|
+
db.set_review_state(conn, plan_id, "reviewed") # BE-D4
|
|
510
|
+
return {
|
|
511
|
+
"ready": True,
|
|
512
|
+
"plan_status": "COMPLETED",
|
|
513
|
+
"review_step_id": review_step_id,
|
|
514
|
+
"review_status": "COMPLETED",
|
|
515
|
+
"review_child_step_id": child["step_id"],
|
|
516
|
+
"reason": "agent review child step COMPLETED; finalizing plan COMPLETED",
|
|
517
|
+
}
|
|
518
|
+
if child["status"] == "FAILED":
|
|
519
|
+
db.update_step_status(conn, review_step_id, "FAILED", set_completed=True)
|
|
520
|
+
db.update_plan_status(conn, plan_id, "FAILED")
|
|
521
|
+
db.set_review_state(conn, plan_id, "reviewed") # BE-D4
|
|
522
|
+
return {
|
|
523
|
+
"ready": True,
|
|
524
|
+
"plan_status": "FAILED",
|
|
525
|
+
"review_step_id": review_step_id,
|
|
526
|
+
"review_status": "FAILED",
|
|
527
|
+
"review_child_step_id": child["step_id"],
|
|
528
|
+
"reason": "agent review child step FAILED; propagating FAILED to plan",
|
|
529
|
+
}
|
|
530
|
+
# child still PENDING / IN_PROGRESS — keep waiting (idempotent no-op)
|
|
531
|
+
return {
|
|
532
|
+
"ready": False,
|
|
533
|
+
"needs_agent_review": True,
|
|
534
|
+
"project": detect_registered_project(conn, plan_id, registry_path),
|
|
535
|
+
"plan_status": "IN_PROGRESS",
|
|
536
|
+
"review_step_id": review_step_id,
|
|
537
|
+
"review_child_step_id": child["step_id"],
|
|
538
|
+
"review_status": "NEEDS_REVIEW",
|
|
539
|
+
"reason": "review step NEEDS_REVIEW; awaiting agent child step outcome",
|
|
540
|
+
}
|
|
541
|
+
|
|
542
|
+
# Tally sibling (non-review) step statuses
|
|
543
|
+
tally_rows = conn.execute(
|
|
544
|
+
"SELECT status, COUNT(*) AS n FROM Steps "
|
|
545
|
+
"WHERE plan_id = ? AND is_review = 0 GROUP BY status",
|
|
546
|
+
(plan_id,),
|
|
547
|
+
).fetchall()
|
|
548
|
+
tally = {row["status"]: row["n"] for row in tally_rows}
|
|
549
|
+
pending = tally.get("PENDING", 0) + tally.get("STARTING", 0)
|
|
550
|
+
in_progress = tally.get("IN_PROGRESS", 0)
|
|
551
|
+
failed = tally.get("FAILED", 0)
|
|
552
|
+
completed = tally.get("COMPLETED", 0)
|
|
553
|
+
total_non_review = pending + in_progress + failed + completed
|
|
554
|
+
|
|
555
|
+
# Decision tree
|
|
556
|
+
if pending > 0 or in_progress > 0:
|
|
557
|
+
return {
|
|
558
|
+
"ready": False,
|
|
559
|
+
"plan_status": "IN_PROGRESS",
|
|
560
|
+
"review_step_id": review_step_id,
|
|
561
|
+
"review_status": "PENDING",
|
|
562
|
+
"reason": (
|
|
563
|
+
f"{pending} pending + {in_progress} in_progress step(s) remaining; "
|
|
564
|
+
f"review deferred"
|
|
565
|
+
),
|
|
566
|
+
}
|
|
567
|
+
|
|
568
|
+
if total_non_review == 0:
|
|
569
|
+
# Plan has zero non-review steps (weird edge case — publish-plan rejects
|
|
570
|
+
# empty steps lists). Treat as COMPLETED for safety.
|
|
571
|
+
new_plan_status = "COMPLETED"
|
|
572
|
+
new_review_status = "COMPLETED"
|
|
573
|
+
reason = "plan has zero non-review steps; trivially complete"
|
|
574
|
+
elif failed > 0:
|
|
575
|
+
# New recovery-aware logic (Jun 2026): a FAILED step is 'recovered' if
|
|
576
|
+
# its deviation sub-tree resolves successfully. Only UNRECOVERED FAILED
|
|
577
|
+
# steps poison the plan.
|
|
578
|
+
failed_step_ids = [
|
|
579
|
+
r["step_id"] for r in conn.execute(
|
|
580
|
+
"SELECT step_id FROM Steps "
|
|
581
|
+
"WHERE plan_id = ? AND is_review = 0 AND status = 'FAILED'",
|
|
582
|
+
(plan_id,),
|
|
583
|
+
).fetchall()
|
|
584
|
+
]
|
|
585
|
+
# Only consider TOP-LEVEL failed steps (parent_step_id IS NULL or parent
|
|
586
|
+
# is itself non-FAILED) — children of a FAILED parent are already
|
|
587
|
+
# accounted for by the recursive _is_step_recovered walk and would
|
|
588
|
+
# otherwise double-count.
|
|
589
|
+
top_failed = []
|
|
590
|
+
for sid in failed_step_ids:
|
|
591
|
+
parent = conn.execute(
|
|
592
|
+
"SELECT parent_step_id FROM Steps WHERE step_id = ?", (sid,)
|
|
593
|
+
).fetchone()
|
|
594
|
+
parent_id = parent["parent_step_id"] if parent else None
|
|
595
|
+
if parent_id is None:
|
|
596
|
+
top_failed.append(sid)
|
|
597
|
+
else:
|
|
598
|
+
# If parent is itself FAILED, this child is part of the parent's
|
|
599
|
+
# recovery sub-tree — don't count it independently.
|
|
600
|
+
p_row = conn.execute(
|
|
601
|
+
"SELECT status FROM Steps WHERE step_id = ?", (parent_id,)
|
|
602
|
+
).fetchone()
|
|
603
|
+
if not p_row or p_row["status"] != "FAILED":
|
|
604
|
+
top_failed.append(sid)
|
|
605
|
+
unrecovered = [sid for sid in top_failed if not _is_step_recovered(conn, sid)]
|
|
606
|
+
if unrecovered:
|
|
607
|
+
new_plan_status = "FAILED"
|
|
608
|
+
new_review_status = "FAILED"
|
|
609
|
+
sample = ", ".join(unrecovered[:3])
|
|
610
|
+
more = f" (+{len(unrecovered)-3} more)" if len(unrecovered) > 3 else ""
|
|
611
|
+
reason = (
|
|
612
|
+
f"{len(unrecovered)} unrecovered failed step(s) [{sample}{more}]; "
|
|
613
|
+
f"propagating FAILED to plan"
|
|
614
|
+
)
|
|
615
|
+
else:
|
|
616
|
+
new_plan_status = "COMPLETED"
|
|
617
|
+
new_review_status = "COMPLETED"
|
|
618
|
+
reason = (
|
|
619
|
+
f"all {len(top_failed)} failure(s) recovered via deviation "
|
|
620
|
+
f"sub-tree(s); finalizing plan as COMPLETED"
|
|
621
|
+
)
|
|
622
|
+
else:
|
|
623
|
+
# All non-review steps COMPLETED
|
|
624
|
+
new_plan_status = "COMPLETED"
|
|
625
|
+
new_review_status = "COMPLETED"
|
|
626
|
+
reason = f"all {completed} non-review step(s) COMPLETED; finalizing plan"
|
|
627
|
+
|
|
628
|
+
# LLM-review routing: when the plan WOULD close COMPLETED and it mentions a
|
|
629
|
+
# registered project, defer the close to an agent review instead. The FAILED
|
|
630
|
+
# path is never intercepted — unrecovered failures propagate immediately.
|
|
631
|
+
if new_plan_status == "COMPLETED":
|
|
632
|
+
project = detect_registered_project(conn, plan_id, registry_path)
|
|
633
|
+
if project is not None:
|
|
634
|
+
child_id = _open_agent_review(conn, plan_id, review_step_id)
|
|
635
|
+
return {
|
|
636
|
+
"ready": False,
|
|
637
|
+
"needs_agent_review": True,
|
|
638
|
+
"project": project,
|
|
639
|
+
"plan_status": "IN_PROGRESS",
|
|
640
|
+
"review_step_id": review_step_id,
|
|
641
|
+
"review_child_step_id": child_id,
|
|
642
|
+
"review_status": "NEEDS_REVIEW",
|
|
643
|
+
"reason": (
|
|
644
|
+
f"plan mentions registered project '{project}'; deferring "
|
|
645
|
+
f"completion to LLM sub-agent review (NEEDS_REVIEW), child "
|
|
646
|
+
f"step {child_id} created"
|
|
647
|
+
),
|
|
648
|
+
}
|
|
649
|
+
|
|
650
|
+
# Apply writes: review step status first, then plan status.
|
|
651
|
+
# Use db.update_step_status with set_completed=True so completed_at gets set.
|
|
652
|
+
db.update_step_status(conn, review_step_id, new_review_status, set_completed=True)
|
|
653
|
+
db.update_plan_status(conn, plan_id, new_plan_status)
|
|
654
|
+
|
|
655
|
+
return {
|
|
656
|
+
"ready": True,
|
|
657
|
+
"plan_status": new_plan_status,
|
|
658
|
+
"review_step_id": review_step_id,
|
|
659
|
+
"review_status": new_review_status,
|
|
660
|
+
"reason": reason,
|
|
661
|
+
}
|