code-factory-2-forge 0.6.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,76 @@
1
+ """The Intent Thread — end-to-end traceability from PRD to production.
2
+
3
+ The breakthrough: SpecLine rationalizes intent into a sealed envelope
4
+ (coherence + surfaced assumptions + hash). ForgeLine consumes that same
5
+ envelope so the FINAL shipped code can be verified against the ORIGINAL
6
+ intent — not the plan, not the spec-as-drifted, but the rationalized intent
7
+ that was sealed at the plan gate.
8
+
9
+ This closes the last translation-loss gap (implementation → validation):
10
+ 'does the shipped thing actually satisfy what we rationalized we wanted?'
11
+ Every surfaced assumption becomes a checkable obligation; the sealed hash
12
+ proves the intent hasn't been quietly swapped underneath the build.
13
+ """
14
+ from __future__ import annotations
15
+ import json, hashlib
16
+ from dataclasses import dataclass, field
17
+ from pathlib import Path
18
+
19
+ @dataclass
20
+ class IntentTraceResult:
21
+ envelope_found: bool = False
22
+ intent_hash: str = ""
23
+ coherence_at_seal: int = 0
24
+ assumptions: list = field(default_factory=list)
25
+ unverified_assumptions: list = field(default_factory=list)
26
+ obligations_met: int = 0
27
+ obligations_total: int = 0
28
+ findings: list = field(default_factory=list)
29
+
30
+ @property
31
+ def traceable(self) -> bool:
32
+ return self.envelope_found and not self.unverified_assumptions
33
+
34
+ def load_envelope(root: Path, feature: str) -> dict | None:
35
+ """Load the SpecLine intent envelope if present (cross-module seam)."""
36
+ for cand in [root/"envelopes"/f"{feature}.json",
37
+ root.parent/"specline"/"envelopes"/f"{feature}.json"]:
38
+ if cand.exists():
39
+ return json.loads(cand.read_text())
40
+ return None
41
+
42
+ def verify_against_intent(root: Path, feature: str, src_dir: Path) -> IntentTraceResult:
43
+ """Check shipped code honors the sealed intent's surfaced assumptions.
44
+ Each assumption becomes an obligation the code must visibly address."""
45
+ r = IntentTraceResult()
46
+ env = load_envelope(root, feature)
47
+ if not env:
48
+ r.findings.append("IT_NO_ENVELOPE: no sealed intent envelope — build is untraceable to a rationalized PRD.")
49
+ return r
50
+ r.envelope_found = True
51
+ r.intent_hash = env.get("sealed_hash", "")
52
+ r.coherence_at_seal = env.get("coherence_score", 0)
53
+ r.assumptions = env.get("assumptions", [])
54
+ r.obligations_total = len(r.assumptions)
55
+
56
+ # each surfaced assumption is an obligation — is it visibly addressed in code?
57
+ code_text = "\n".join(p.read_text() for p in Path(src_dir).rglob("*.py")).lower()
58
+ OBLIGATION_EVIDENCE = {
59
+ "auth": ["auth", "login", "token", "session", "identity", "authenticate"],
60
+ "currency": ["currency", "usd", "locale", "money", "decimal"],
61
+ "dependency":["retry", "timeout", "fallback", "except", "circuit", "backoff"],
62
+ "delivery": ["retry", "bounce", "queue", "deliver", "ack", "confirm"],
63
+ }
64
+ for a in r.assumptions:
65
+ al = a.lower()
66
+ key = ("auth" if "auth" in al or "identity" in al else
67
+ "currency" if "currency" in al else
68
+ "dependency" if "dependency" in al or "availability" in al else
69
+ "delivery" if "delivery" in al else None)
70
+ evidence = OBLIGATION_EVIDENCE.get(key, [])
71
+ if evidence and any(e in code_text for e in evidence):
72
+ r.obligations_met += 1
73
+ else:
74
+ r.unverified_assumptions.append(a)
75
+ r.findings.append(f"IT_UNMET_ASSUMPTION: intent assumed — '{a[:70]}' — but code shows no handling.")
76
+ return r
forgeline/learning.py ADDED
@@ -0,0 +1,116 @@
1
+ """Recursive Learning Kernel — the flywheel that makes the factory improve
2
+ itself. Distinct from skill_memory (which records lessons): this CLOSES THE
3
+ LOOP by promoting recurring lessons into ACTIVE enforcement, tracking whether
4
+ promoted rules actually reduce future failures, and DEMOTING rules that stop
5
+ earning their keep. The factory's own run history becomes its QA policy.
6
+
7
+ Three tiers, each a ratchet:
8
+ observe -> a lesson is recorded (skill_memory)
9
+ promote -> a lesson seen >= N times becomes an ACTIVE constraint (enforced)
10
+ validate -> a promoted constraint that prevents recurrence is KEPT; one that
11
+ never fires again (dead) or fires with no failures behind it is
12
+ reviewed. Effectiveness is measured, not assumed.
13
+ """
14
+ from __future__ import annotations
15
+ import json, datetime
16
+ from dataclasses import dataclass, asdict
17
+ from pathlib import Path
18
+
19
+ POLICY = "skills/active_policy.json"
20
+ HISTORY = "skills/learning_history.jsonl"
21
+
22
+ def _now(): return datetime.datetime.now(datetime.timezone.utc).isoformat()
23
+
24
+ @dataclass
25
+ class ActiveConstraint:
26
+ code: str # e.g. "A_EVAL"
27
+ rule: str # human-readable enforced rule
28
+ promoted_from_count: int # how many observations triggered promotion
29
+ promoted_at: str
30
+ times_enforced: int = 0 # how often it has since blocked something
31
+ prevented_recurrence: int = 0 # times it caught the same failure again
32
+ status: str = "active" # active | probation | retired
33
+
34
+ class LearningKernel:
35
+ def __init__(self, root: Path, promote_threshold: int = 3):
36
+ self.root = Path(root)
37
+ self.threshold = promote_threshold
38
+ self.policy_path = self.root/POLICY
39
+ self.history_path = self.root/HISTORY
40
+ self.policy_path.parent.mkdir(parents=True, exist_ok=True)
41
+
42
+ def _load_policy(self) -> dict:
43
+ if self.policy_path.exists():
44
+ return json.loads(self.policy_path.read_text())
45
+ return {"constraints": {}, "version": 0}
46
+
47
+ def _save_policy(self, p: dict):
48
+ p["version"] = p.get("version", 0) + 1
49
+ p["updated"] = _now()
50
+ self.policy_path.write_text(json.dumps(p, indent=2))
51
+
52
+ def _log(self, event: str, **f):
53
+ f.update({"ts": _now(), "event": event})
54
+ with self.history_path.open("a") as fh:
55
+ fh.write(json.dumps(f, sort_keys=True) + "\n")
56
+
57
+ def promote(self, lessons: list[dict]) -> list[str]:
58
+ """Tier 2: any lesson seen >= threshold becomes an active constraint.
59
+ Returns the codes newly promoted this cycle."""
60
+ policy = self._load_policy(); newly = []
61
+ for l in lessons:
62
+ code = l["failure_code"]
63
+ if l["count"] >= self.threshold and code not in policy["constraints"]:
64
+ c = ActiveConstraint(code=code, rule=l["fix"],
65
+ promoted_from_count=l["count"], promoted_at=_now())
66
+ policy["constraints"][code] = asdict(c)
67
+ newly.append(code)
68
+ self._log("promote", code=code, from_count=l["count"], rule=l["fix"])
69
+ if newly:
70
+ self._save_policy(policy)
71
+ return newly
72
+
73
+ def enforce(self, findings_codes: list[str]) -> dict:
74
+ """Tier 3 measurement: record which active constraints fired this run.
75
+ A constraint that catches its target failure again = validated (working).
76
+ Returns {code: prevented_bool} for constraints that were relevant."""
77
+ policy = self._load_policy(); result = {}
78
+ for code in findings_codes:
79
+ if code in policy["constraints"]:
80
+ c = policy["constraints"][code]
81
+ c["times_enforced"] += 1
82
+ c["prevented_recurrence"] += 1
83
+ result[code] = True
84
+ self._log("enforce", code=code, prevented=True)
85
+ if result:
86
+ self._save_policy(policy)
87
+ return result
88
+
89
+ def audit_effectiveness(self, recent_run_count: int = 20) -> dict:
90
+ """Recursive review: constraints that keep catching real failures are
91
+ VALIDATED; ones promoted long ago that never fire go to PROBATION
92
+ (candidate for retirement — the policy self-prunes)."""
93
+ policy = self._load_policy()
94
+ validated, probation = [], []
95
+ for code, c in policy["constraints"].items():
96
+ if c["status"] == "retired":
97
+ continue
98
+ if c["prevented_recurrence"] >= 1:
99
+ c["status"] = "active"; validated.append(code)
100
+ elif c["times_enforced"] == 0 and c["status"] == "active":
101
+ c["status"] = "probation"; probation.append(code)
102
+ self._log("probation", code=code, reason="never fired since promotion")
103
+ self._save_policy(policy)
104
+ return {"validated": validated, "probation": probation,
105
+ "total_active": sum(1 for c in policy["constraints"].values() if c["status"]=="active")}
106
+
107
+ def active_codes(self) -> list[str]:
108
+ policy = self._load_policy()
109
+ return [k for k,v in policy["constraints"].items() if v["status"] in ("active","probation")]
110
+
111
+ def policy_summary(self) -> dict:
112
+ policy = self._load_policy()
113
+ return {"version": policy.get("version",0),
114
+ "constraints": {k: {"status": v["status"], "enforced": v["times_enforced"],
115
+ "prevented": v["prevented_recurrence"]}
116
+ for k,v in policy["constraints"].items()}}
@@ -0,0 +1,263 @@
1
+ """The outer loop. Advances the state machine, calling SpecLine for spec/plan
2
+ governance and HSF for decision compilation when a spec carries a decision
3
+ table. Everything is a receipt; every gate failure records a skill lesson and
4
+ routes to the refine loop."""
5
+ from __future__ import annotations
6
+ import shutil, subprocess, sys
7
+ from pathlib import Path
8
+ from .states import State, can_transition, IllegalTransition, HUMAN_GATES
9
+ from .run_store import RunStore
10
+ from .ssat import load_ssat, scaffold_from_ssat, check_erosion
11
+ from .gates import judge_consistency, grumpy_review, skill_check
12
+ from .gates.qa_audit import qa_audit
13
+ from .skill_memory import record_lesson, inject_lessons_block, lessons_for
14
+ from .learning import LearningKernel
15
+ from .attribution import Attribution, FailureClass, UnitResult
16
+
17
+ MAX_REFINE = 3
18
+
19
+
20
+ def _gate_attribution(stage: str, checks: list[tuple[str, bool, str, FailureClass]]) -> dict:
21
+ units = [
22
+ UnitResult(
23
+ unit=f"{stage}:{name}",
24
+ stage=stage,
25
+ passed=passed,
26
+ evidence=evidence,
27
+ failure_class=None if passed else failure_class,
28
+ )
29
+ for name, passed, evidence, failure_class in checks
30
+ ]
31
+ return Attribution(stage, len(units), sum(unit.passed for unit in units), units).to_dict()
32
+
33
+ class Orchestrator:
34
+ def __init__(self, root: Path, feature: str):
35
+ self.root = Path(root); self.feature = feature
36
+ self.store = RunStore(self.root, feature)
37
+
38
+ def _advance(self, to: State, note: str = "", **receipt):
39
+ frm = self.store.state
40
+ if not can_transition(frm, to):
41
+ raise IllegalTransition(f"{frm.value} -> {to.value}")
42
+ self.store.set_state(to, note)
43
+ self.store.receipt(transition=f"{frm.value}->{to.value}", **receipt)
44
+
45
+ def architect(self, ssat_path: Path) -> dict:
46
+ ssat = load_ssat(ssat_path)
47
+ created = scaffold_from_ssat(ssat, self.root)
48
+ self._advance(State.SCAFFOLDED, "scaffold from SSAT",
49
+ modules=len(ssat.get("modules", [])), files=[str(c) for c in created])
50
+ return {"scaffolded": [str(c) for c in created]}
51
+
52
+ def review(self, ssat_path: Path) -> dict:
53
+ """Judge + grumpy adversary + arch erosion + deep QA audit, with a
54
+ recursive learning kernel and escalating refine loop."""
55
+ ssat = load_ssat(ssat_path)
56
+ attempt = self.store.bump_attempt("review")
57
+ kernel = LearningKernel(self.root)
58
+ j_ok, j = judge_consistency(ssat, self.root)
59
+ g_ok, g = grumpy_review(self.root)
60
+ erosion = check_erosion(ssat, self.root)
61
+ qa = qa_audit(self.root) # STRICTER: quantitative QA grade
62
+ all_findings = j + g + [f"{v.code} {v.message}" for v in erosion] + qa.findings
63
+ all_ok = j_ok and g_ok and not erosion and qa.passed
64
+ review_attr = _gate_attribution("review", [
65
+ ("judge", j_ok, "consistent" if j_ok else " | ".join(j),
66
+ FailureClass.INCONSISTENT_LOGIC),
67
+ ("adversary", g_ok, "all probes resisted" if g_ok else " | ".join(g),
68
+ FailureClass.SECURITY_FINDING),
69
+ ("architecture", not erosion, "no erosion" if not erosion else
70
+ " | ".join(f"{v.code} {v.message}" for v in erosion),
71
+ FailureClass.SIGNATURE_DRIFT),
72
+ ("qa_audit", qa.passed, f"grade={qa.grade}; metrics={qa.metrics}",
73
+ FailureClass.COMPLEXITY_EXCEEDED),
74
+ ])
75
+
76
+ # recursive learning: measure whether active constraints caught failures
77
+ codes = [f.split()[0].split("[")[0] for f in all_findings]
78
+ prevented = kernel.enforce(codes)
79
+
80
+ self.store.receipt(phase="review", attempt=attempt, judge_ok=j_ok,
81
+ grumpy_ok=g_ok, erosion=len(erosion),
82
+ qa_grade=qa.grade, qa_metrics=qa.metrics,
83
+ active_constraints_fired=list(prevented.keys()),
84
+ findings=all_findings, attribution=review_attr)
85
+ if all_ok:
86
+ self._advance(State.REVIEWED, f"review pass (attempt {attempt}, QA={qa.grade})")
87
+ return {"reviewed": True, "attempt": attempt, "qa_grade": qa.grade,
88
+ "qa_metrics": qa.metrics, "attribution": review_attr}
89
+
90
+ # record lessons, then PROMOTE recurring ones into active policy (recursion)
91
+ for f in all_findings:
92
+ code = f.split()[0].split("[")[0]
93
+ record_lesson(self.root, phase="fill", failure_code=code, fix=f, feature=self.feature)
94
+ promoted = kernel.promote(lessons_for(self.root, "fill"))
95
+ kernel.audit_effectiveness()
96
+
97
+ if self.store.state != State.BLOCKED:
98
+ self._advance(State.BLOCKED, f"review failed (attempt {attempt}, QA={qa.grade})")
99
+
100
+ # ESCALATION: refine loop gets stricter each attempt
101
+ escalation = ("normal" if attempt == 1 else
102
+ "elevated: fix ALL findings, not just blockers" if attempt == 2 else
103
+ "final: exhausted — human review required" )
104
+ result = {"reviewed": False, "attempt": attempt, "qa_grade": qa.grade,
105
+ "qa_metrics": qa.metrics, "findings": all_findings,
106
+ "newly_promoted_constraints": promoted, "escalation": escalation,
107
+ "lessons_for_next": inject_lessons_block(self.root, "fill"),
108
+ "attribution": review_attr}
109
+ if attempt >= MAX_REFINE:
110
+ result["exhausted"] = True
111
+ return result
112
+
113
+ def arch_gate(self, ssat_path: Path) -> dict:
114
+ ssat = load_ssat(ssat_path)
115
+ erosion = check_erosion(ssat, self.root)
116
+ s_ok, s = skill_check(self.root, self.feature)
117
+ arch_attr = _gate_attribution("arch_gate", [
118
+ ("erosion", not erosion, "no architecture erosion" if not erosion else
119
+ " | ".join(f"{v.code} {v.message}" for v in erosion),
120
+ FailureClass.SIGNATURE_DRIFT),
121
+ ("skill_contract", s_ok, "skill contract valid" if s_ok else " | ".join(s),
122
+ FailureClass.INCONSISTENT_LOGIC),
123
+ ])
124
+ if erosion or not s_ok:
125
+ self.store.receipt(phase="arch_gate", passed=False,
126
+ erosion=[f"{v.code} {v.message}" for v in erosion], skill=s,
127
+ attribution=arch_attr)
128
+ if self.store.state != State.BLOCKED:
129
+ self._advance(State.BLOCKED, "arch gate failed")
130
+ return {"passed": False, "erosion": [f"{v.code} {v.message}" for v in erosion] + s,
131
+ "attribution": arch_attr}
132
+ # must currently be REVIEWED to legally gate
133
+ if self.store.state == State.BLOCKED:
134
+ self.store.set_state(State.REVIEWED, "recovered for arch gate")
135
+ kernel = LearningKernel(self.root)
136
+ eff = kernel.audit_effectiveness()
137
+ self._advance(State.ARCH_GATED, "architecture CI gate passed")
138
+ self.store.receipt(phase="arch_gate", passed=True, learning=eff,
139
+ active_policy=kernel.policy_summary(), attribution=arch_attr)
140
+ return {"passed": True, "learning": eff, "attribution": arch_attr}
141
+
142
+ def handoff_decisions(self, spec_path: Path) -> dict | None:
143
+ """If SpecLine + HSF are importable and the spec has a decision table,
144
+ compile decisions through the factory. Best-effort seam."""
145
+ try:
146
+ sys.path.insert(0, str(Path(__file__).resolve().parents[2]/"specline"))
147
+ from specline.spec_lint import decision_rows # type: ignore
148
+ except Exception:
149
+ return None
150
+ rows = decision_rows(spec_path)
151
+ if not rows:
152
+ return None
153
+ self.store.receipt(phase="handoff", decision_rules=len(rows),
154
+ note="decision table present — route to HSF compile")
155
+ return {"decision_rules": len(rows), "next": "hsf compile"}
156
+
157
+ def smoke_gate(self) -> dict:
158
+ """Runtime behavior verification — the behavior-by-inspection gate.
159
+ Runs the built artifact against declared behavioral checks (smoke/<feature>.json)
160
+ in isolated subprocesses. Blocks ship on any runtime failure. This is the
161
+ right-sized version of a per-PR preview deployment: verify the thing RUNS
162
+ and behaves, not just that it type-checks and honors the spec."""
163
+ from .gates.runtime_smoke import runtime_smoke, smoke_report_lines
164
+ if self.store.state == State.SHIPPED:
165
+ return {"smoked": True, "note": "already shipped"}
166
+ if self.store.state == State.SMOKED:
167
+ return {"smoked": True, "note": "already smoked"}
168
+ if self.store.state != State.TESTS_VERIFIED:
169
+ return {"smoked": False,
170
+ "reason": "tests not verified - run `forge verify-tests` first"}
171
+ rep = runtime_smoke(self.root, self.feature)
172
+ gate = rep.gate_result
173
+ attr = gate.attribution.to_dict()
174
+ lines = smoke_report_lines(rep)
175
+ if not gate.passed:
176
+ self.store.receipt(phase="smoke_gate", smoked=False,
177
+ manifest_found=rep.manifest_found,
178
+ failures=[r.name for r in rep.failures],
179
+ report=lines, attribution=attr)
180
+ # a runtime failure is a real defect -> block; refine loop owns recovery
181
+ self._advance(State.BLOCKED, "runtime smoke gate failed")
182
+ return {"smoked": False,
183
+ "reason": ("no runtime behavior verified" if not rep.manifest_found
184
+ else "runtime behavioral check(s) failed"),
185
+ "failures": [f"{r.name}: {r.reason}" for r in rep.failures],
186
+ "attribution": attr}
187
+ self._advance(State.SMOKED, "runtime behavior verified")
188
+ self.store.receipt(phase="smoke_gate", smoked=True,
189
+ checks=len(rep.results), report=lines, attribution=attr)
190
+ return {"smoked": True, "checks": len(rep.results), "attribution": attr}
191
+
192
+ def verify_tests(self, ssat_path: Path) -> dict:
193
+ """Verify smoke checks can fail before trusting smoke results."""
194
+ from .gates.reverse_classical import verify_tests
195
+ if self.store.state == State.SHIPPED:
196
+ return {"verified": True, "note": "already shipped"}
197
+ if self.store.state in {State.TESTS_VERIFIED, State.SMOKED}:
198
+ return {"verified": True, "note": "already verified"}
199
+ if self.store.state not in {State.ARCH_GATED, State.BLOCKED}:
200
+ return {"verified": False,
201
+ "reason": "architecture gate not passed - run `forge arch-gate` first"}
202
+ gate = verify_tests(self.root, self.feature, Path(ssat_path))
203
+ attr = gate.attribution.to_dict()
204
+ if not gate.passed:
205
+ self.store.receipt(phase="verify_tests", verified=False, attribution=attr)
206
+ if self.store.state != State.BLOCKED:
207
+ self._advance(State.BLOCKED, "reverse-classical test verification failed")
208
+ return {"verified": False,
209
+ "reason": "hollow smoke check(s) found",
210
+ "attribution": attr}
211
+ self._advance(State.TESTS_VERIFIED, "smoke checks proven non-hollow",
212
+ attribution=attr)
213
+ self.store.receipt(phase="verify_tests", verified=True, attribution=attr)
214
+ return {"verified": True, "checks": gate.attribution.n_checked, "attribution": attr}
215
+
216
+ def refine(self, evaluate, propose, apply, revert, max_iters: int = 6) -> dict:
217
+ """Run deterministic localized refinement.
218
+
219
+ Callers may propose an edit, but exact Pareto acceptance and rollback are
220
+ controlled here. Learning state remains build-time only under `.forge/`.
221
+ """
222
+ from .refinement import refine
223
+ return refine(evaluate, propose, apply, revert, self.root, max_iters=max_iters)
224
+
225
+ def ship(self, verify_intent: bool = True) -> dict:
226
+ # ship now requires runtime behavior to have been verified
227
+ if self.store.state in {State.ARCH_GATED, State.TESTS_VERIFIED}:
228
+ return {"shipped": False, "reason": "runtime smoke gate not run - "
229
+ "call smoke_gate() before ship (behavior must be verified)."}
230
+ trace = None
231
+ intent_attr = None
232
+ if verify_intent:
233
+ from .intent_thread import verify_against_intent
234
+ trace = verify_against_intent(self.root, self.feature, self.root)
235
+ intent_checks = [
236
+ (f"obligation_{index}", met, evidence, FailureClass.INCONSISTENT_LOGIC)
237
+ for index, (met, evidence) in enumerate(
238
+ [(not trace.unverified_assumptions, finding)
239
+ for finding in (trace.findings or ["intent obligations satisfied"])],
240
+ 1,
241
+ )
242
+ ]
243
+ intent_attr = _gate_attribution("intent_thread", intent_checks)
244
+ if trace.envelope_found and not trace.traceable:
245
+ # PRD->production gap: shipped code doesn't honor sealed intent
246
+ self.store.receipt(phase="ship", shipped=False,
247
+ intent_hash=trace.intent_hash,
248
+ unverified_assumptions=trace.unverified_assumptions,
249
+ findings=trace.findings, attribution=intent_attr)
250
+ if self.store.state != State.BLOCKED:
251
+ self._advance(State.BLOCKED, "intent-traceability gap")
252
+ return {"shipped": False, "reason": "intent not honored by code",
253
+ "unverified_assumptions": trace.unverified_assumptions,
254
+ "findings": trace.findings}
255
+ self._advance(State.SHIPPED, "all gates green")
256
+ self.store.receipt(phase="ship", shipped=True,
257
+ intent_hash=(trace.intent_hash if trace else None),
258
+ obligations=f"{trace.obligations_met}/{trace.obligations_total}" if trace else None,
259
+ attribution=intent_attr)
260
+ return {"shipped": True,
261
+ "intent_traceable": (trace.traceable if trace else None),
262
+ "obligations_met": (f"{trace.obligations_met}/{trace.obligations_total}" if trace else None),
263
+ "attribution": intent_attr}
@@ -0,0 +1,76 @@
1
+ """Deterministic refinement policy: one edit, ordered, Pareto-gated, bounded."""
2
+ from __future__ import annotations
3
+ from dataclasses import asdict, dataclass
4
+ from pathlib import Path
5
+ import json
6
+
7
+ from .attribution import FailureClass
8
+
9
+
10
+ @dataclass(frozen=True)
11
+ class Edit:
12
+ edit_class: str
13
+ target: str
14
+ description: str
15
+
16
+
17
+ STRUCTURAL_FAILURES = {
18
+ FailureClass.SIGNATURE_DRIFT,
19
+ FailureClass.STUB_UNFILLED,
20
+ FailureClass.INCONSISTENT_LOGIC,
21
+ FailureClass.SCOPE_ESCAPE,
22
+ FailureClass.HOLLOW_TEST,
23
+ FailureClass.HOLLOW_MANIFEST,
24
+ }
25
+
26
+
27
+ def select_edit(failure_class: FailureClass) -> Edit:
28
+ edit_class = "structural" if failure_class in STRUCTURAL_FAILURES else (
29
+ "configuration" if failure_class in {
30
+ FailureClass.RUNTIME_CRASH, FailureClass.RUNTIME_TIMEOUT,
31
+ FailureClass.SECURITY_FINDING,
32
+ } else "parametric"
33
+ )
34
+ return Edit(edit_class, failure_class.value, f"localized {edit_class} correction")
35
+
36
+
37
+ def pareto_win(current: dict[str, float], previous: dict[str, float], target: str) -> bool:
38
+ return (
39
+ current.get(target, 0.0) > previous.get(target, 0.0)
40
+ and all(current.get(stage, 0.0) >= rate for stage, rate in previous.items())
41
+ )
42
+
43
+
44
+ class RejectionLedger:
45
+ def __init__(self, root: Path):
46
+ self.path = Path(root) / ".forge" / "rejection_ledger.jsonl"
47
+
48
+ def log(self, edit: Edit, before: dict, after: dict) -> None:
49
+ self.path.parent.mkdir(parents=True, exist_ok=True)
50
+ with self.path.open("a", encoding="utf-8") as stream:
51
+ stream.write(json.dumps({
52
+ "edit": asdict(edit), "before_rates": before, "after_rates": after,
53
+ }, sort_keys=True) + "\n")
54
+
55
+
56
+ def refine(evaluate, propose, apply, revert, root: Path, max_iters: int = 6) -> dict:
57
+ previous = evaluate()
58
+ no_win_streak = 0
59
+ ledger = RejectionLedger(root)
60
+ for iteration in range(max_iters):
61
+ if all(rate == 1.0 for rate in previous.values()):
62
+ return {"converged": True, "iters": iteration}
63
+ target, failure_class = propose(previous)
64
+ edit = select_edit(failure_class)
65
+ snapshot = apply(edit)
66
+ current = evaluate()
67
+ if pareto_win(current, previous, target):
68
+ previous = current
69
+ no_win_streak = 0
70
+ continue
71
+ revert(snapshot)
72
+ ledger.log(edit, previous, current)
73
+ no_win_streak += 1
74
+ if no_win_streak >= 2:
75
+ return {"converged": False, "reason": "plateau", "iters": iteration + 1}
76
+ return {"converged": False, "reason": "budget", "iters": max_iters}
forgeline/run_store.py ADDED
@@ -0,0 +1,43 @@
1
+ """Per-feature run state + receipts. The state machine's durable memory —
2
+ survives context resets (Ralph Wiggum) because the disk is the truth."""
3
+ from __future__ import annotations
4
+ import json, datetime, hashlib
5
+ from pathlib import Path
6
+ from .states import State
7
+
8
+ def _now():
9
+ return datetime.datetime.now(datetime.timezone.utc).isoformat()
10
+
11
+ class RunStore:
12
+ def __init__(self, root: Path, feature: str):
13
+ self.root = Path(root); self.feature = feature
14
+ self.dir = self.root/".forge"/feature
15
+ self.dir.mkdir(parents=True, exist_ok=True)
16
+ self.state_path = self.dir/"state.json"
17
+ self.receipts = self.dir/"receipts.jsonl"
18
+ if not self.state_path.exists():
19
+ self._write({"feature": feature, "state": State.INTENT.value,
20
+ "created": _now(), "attempts": {}, "history": []})
21
+
22
+ def _write(self, data): self.state_path.write_text(json.dumps(data, indent=2))
23
+ def load(self) -> dict: return json.loads(self.state_path.read_text())
24
+
25
+ @property
26
+ def state(self) -> State:
27
+ return State(self.load()["state"])
28
+
29
+ def set_state(self, s: State, note: str = ""):
30
+ d = self.load(); d["state"] = s.value
31
+ d["history"].append({"ts": _now(), "state": s.value, "note": note})
32
+ self._write(d)
33
+
34
+ def bump_attempt(self, phase: str) -> int:
35
+ d = self.load(); d["attempts"][phase] = d["attempts"].get(phase, 0) + 1
36
+ self._write(d); return d["attempts"][phase]
37
+
38
+ def receipt(self, **fields):
39
+ fields["ts"] = _now()
40
+ line = json.dumps(fields, sort_keys=True)
41
+ fields = {"h": hashlib.sha256(line.encode()).hexdigest()[:12], **fields}
42
+ with self.receipts.open("a") as f:
43
+ f.write(json.dumps(fields, sort_keys=True) + "\n")
@@ -0,0 +1,52 @@
1
+ """Skill memory — the self-accelerating flywheel. Records what failed in each
2
+ run and the fix, so future runs inject accumulated lessons into agent context.
3
+ This is the 'skill learns from past refinements' mechanism, made concrete and
4
+ LLM-free: lessons are structured, deduped, and promotable to constraints."""
5
+ from __future__ import annotations
6
+ import json, datetime
7
+ from pathlib import Path
8
+
9
+ LESSONS = "skills/lessons.jsonl"
10
+
11
+ def record_lesson(root: Path, *, phase: str, failure_code: str, fix: str, feature: str):
12
+ root = Path(root); lp = root/LESSONS
13
+ lp.parent.mkdir(parents=True, exist_ok=True)
14
+ # dedupe by (phase, failure_code, fix)
15
+ existing = []
16
+ if lp.exists():
17
+ existing = [json.loads(l) for l in lp.read_text().splitlines() if l.strip()]
18
+ key = (phase, failure_code, fix)
19
+ for e in existing:
20
+ if (e["phase"], e["failure_code"], e["fix"]) == key:
21
+ e["count"] += 1; e["last_seen"] = datetime.datetime.now(datetime.timezone.utc).isoformat()
22
+ lp.write_text("\n".join(json.dumps(x) for x in existing) + "\n")
23
+ return e
24
+ entry = {"phase": phase, "failure_code": failure_code, "fix": fix, "feature": feature,
25
+ "count": 1, "last_seen": datetime.datetime.now(datetime.timezone.utc).isoformat()}
26
+ existing.append(entry)
27
+ lp.write_text("\n".join(json.dumps(x) for x in existing) + "\n")
28
+ return entry
29
+
30
+ def lessons_for(root: Path, phase: str, min_count: int = 1) -> list[dict]:
31
+ lp = Path(root)/LESSONS
32
+ if not lp.exists(): return []
33
+ rows = [json.loads(l) for l in lp.read_text().splitlines() if l.strip()]
34
+ return sorted([r for r in rows if r["phase"] == phase and r["count"] >= min_count],
35
+ key=lambda r: -r["count"])
36
+
37
+ def promotable_constraints(root: Path, threshold: int = 3) -> list[dict]:
38
+ """Lessons seen >= threshold times graduate into hard constraints
39
+ (conventions-into-constraints). These get injected as SSAT invariants."""
40
+ lp = Path(root)/LESSONS
41
+ if not lp.exists(): return []
42
+ rows = [json.loads(l) for l in lp.read_text().splitlines() if l.strip()]
43
+ return [r for r in rows if r["count"] >= threshold]
44
+
45
+ def inject_lessons_block(root: Path, phase: str) -> str:
46
+ """Text block injected into an agent's task context for this phase."""
47
+ ls = lessons_for(root, phase)
48
+ if not ls: return ""
49
+ lines = [f"## Lessons from past runs (phase: {phase})"]
50
+ for l in ls[:8]:
51
+ lines.append(f"- [{l['failure_code']} ×{l['count']}] {l['fix']}")
52
+ return "\n".join(lines)