code-factory-2-forge 0.6.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- code_factory_2_forge-0.6.0.dist-info/METADATA +308 -0
- code_factory_2_forge-0.6.0.dist-info/RECORD +29 -0
- code_factory_2_forge-0.6.0.dist-info/WHEEL +5 -0
- code_factory_2_forge-0.6.0.dist-info/entry_points.txt +2 -0
- code_factory_2_forge-0.6.0.dist-info/licenses/LICENSE-APACHE +18 -0
- code_factory_2_forge-0.6.0.dist-info/licenses/LICENSE-MIT +21 -0
- code_factory_2_forge-0.6.0.dist-info/licenses/NOTICE +6 -0
- code_factory_2_forge-0.6.0.dist-info/top_level.txt +1 -0
- forgeline/__init__.py +14 -0
- forgeline/adapters.py +56 -0
- forgeline/attribution.py +71 -0
- forgeline/cli.py +101 -0
- forgeline/demo.py +83 -0
- forgeline/demo_learning.py +44 -0
- forgeline/gates/__init__.py +4 -0
- forgeline/gates/adversary.py +45 -0
- forgeline/gates/judge.py +26 -0
- forgeline/gates/qa_audit.py +134 -0
- forgeline/gates/reverse_classical.py +118 -0
- forgeline/gates/runtime_smoke.py +194 -0
- forgeline/gates/skill_check.py +14 -0
- forgeline/intent_thread.py +76 -0
- forgeline/learning.py +116 -0
- forgeline/orchestrator.py +263 -0
- forgeline/refinement.py +76 -0
- forgeline/run_store.py +43 -0
- forgeline/skill_memory.py +52 -0
- forgeline/ssat.py +93 -0
- forgeline/states.py +41 -0
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
"""The Intent Thread — end-to-end traceability from PRD to production.
|
|
2
|
+
|
|
3
|
+
The breakthrough: SpecLine rationalizes intent into a sealed envelope
|
|
4
|
+
(coherence + surfaced assumptions + hash). ForgeLine consumes that same
|
|
5
|
+
envelope so the FINAL shipped code can be verified against the ORIGINAL
|
|
6
|
+
intent — not the plan, not the spec-as-drifted, but the rationalized intent
|
|
7
|
+
that was sealed at the plan gate.
|
|
8
|
+
|
|
9
|
+
This closes the last translation-loss gap (implementation → validation):
|
|
10
|
+
'does the shipped thing actually satisfy what we rationalized we wanted?'
|
|
11
|
+
Every surfaced assumption becomes a checkable obligation; the sealed hash
|
|
12
|
+
proves the intent hasn't been quietly swapped underneath the build.
|
|
13
|
+
"""
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
import json, hashlib
|
|
16
|
+
from dataclasses import dataclass, field
|
|
17
|
+
from pathlib import Path
|
|
18
|
+
|
|
19
|
+
@dataclass
|
|
20
|
+
class IntentTraceResult:
|
|
21
|
+
envelope_found: bool = False
|
|
22
|
+
intent_hash: str = ""
|
|
23
|
+
coherence_at_seal: int = 0
|
|
24
|
+
assumptions: list = field(default_factory=list)
|
|
25
|
+
unverified_assumptions: list = field(default_factory=list)
|
|
26
|
+
obligations_met: int = 0
|
|
27
|
+
obligations_total: int = 0
|
|
28
|
+
findings: list = field(default_factory=list)
|
|
29
|
+
|
|
30
|
+
@property
|
|
31
|
+
def traceable(self) -> bool:
|
|
32
|
+
return self.envelope_found and not self.unverified_assumptions
|
|
33
|
+
|
|
34
|
+
def load_envelope(root: Path, feature: str) -> dict | None:
|
|
35
|
+
"""Load the SpecLine intent envelope if present (cross-module seam)."""
|
|
36
|
+
for cand in [root/"envelopes"/f"{feature}.json",
|
|
37
|
+
root.parent/"specline"/"envelopes"/f"{feature}.json"]:
|
|
38
|
+
if cand.exists():
|
|
39
|
+
return json.loads(cand.read_text())
|
|
40
|
+
return None
|
|
41
|
+
|
|
42
|
+
def verify_against_intent(root: Path, feature: str, src_dir: Path) -> IntentTraceResult:
|
|
43
|
+
"""Check shipped code honors the sealed intent's surfaced assumptions.
|
|
44
|
+
Each assumption becomes an obligation the code must visibly address."""
|
|
45
|
+
r = IntentTraceResult()
|
|
46
|
+
env = load_envelope(root, feature)
|
|
47
|
+
if not env:
|
|
48
|
+
r.findings.append("IT_NO_ENVELOPE: no sealed intent envelope — build is untraceable to a rationalized PRD.")
|
|
49
|
+
return r
|
|
50
|
+
r.envelope_found = True
|
|
51
|
+
r.intent_hash = env.get("sealed_hash", "")
|
|
52
|
+
r.coherence_at_seal = env.get("coherence_score", 0)
|
|
53
|
+
r.assumptions = env.get("assumptions", [])
|
|
54
|
+
r.obligations_total = len(r.assumptions)
|
|
55
|
+
|
|
56
|
+
# each surfaced assumption is an obligation — is it visibly addressed in code?
|
|
57
|
+
code_text = "\n".join(p.read_text() for p in Path(src_dir).rglob("*.py")).lower()
|
|
58
|
+
OBLIGATION_EVIDENCE = {
|
|
59
|
+
"auth": ["auth", "login", "token", "session", "identity", "authenticate"],
|
|
60
|
+
"currency": ["currency", "usd", "locale", "money", "decimal"],
|
|
61
|
+
"dependency":["retry", "timeout", "fallback", "except", "circuit", "backoff"],
|
|
62
|
+
"delivery": ["retry", "bounce", "queue", "deliver", "ack", "confirm"],
|
|
63
|
+
}
|
|
64
|
+
for a in r.assumptions:
|
|
65
|
+
al = a.lower()
|
|
66
|
+
key = ("auth" if "auth" in al or "identity" in al else
|
|
67
|
+
"currency" if "currency" in al else
|
|
68
|
+
"dependency" if "dependency" in al or "availability" in al else
|
|
69
|
+
"delivery" if "delivery" in al else None)
|
|
70
|
+
evidence = OBLIGATION_EVIDENCE.get(key, [])
|
|
71
|
+
if evidence and any(e in code_text for e in evidence):
|
|
72
|
+
r.obligations_met += 1
|
|
73
|
+
else:
|
|
74
|
+
r.unverified_assumptions.append(a)
|
|
75
|
+
r.findings.append(f"IT_UNMET_ASSUMPTION: intent assumed — '{a[:70]}' — but code shows no handling.")
|
|
76
|
+
return r
|
forgeline/learning.py
ADDED
|
@@ -0,0 +1,116 @@
|
|
|
1
|
+
"""Recursive Learning Kernel — the flywheel that makes the factory improve
|
|
2
|
+
itself. Distinct from skill_memory (which records lessons): this CLOSES THE
|
|
3
|
+
LOOP by promoting recurring lessons into ACTIVE enforcement, tracking whether
|
|
4
|
+
promoted rules actually reduce future failures, and DEMOTING rules that stop
|
|
5
|
+
earning their keep. The factory's own run history becomes its QA policy.
|
|
6
|
+
|
|
7
|
+
Three tiers, each a ratchet:
|
|
8
|
+
observe -> a lesson is recorded (skill_memory)
|
|
9
|
+
promote -> a lesson seen >= N times becomes an ACTIVE constraint (enforced)
|
|
10
|
+
validate -> a promoted constraint that prevents recurrence is KEPT; one that
|
|
11
|
+
never fires again (dead) or fires with no failures behind it is
|
|
12
|
+
reviewed. Effectiveness is measured, not assumed.
|
|
13
|
+
"""
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
import json, datetime
|
|
16
|
+
from dataclasses import dataclass, asdict
|
|
17
|
+
from pathlib import Path
|
|
18
|
+
|
|
19
|
+
POLICY = "skills/active_policy.json"
|
|
20
|
+
HISTORY = "skills/learning_history.jsonl"
|
|
21
|
+
|
|
22
|
+
def _now(): return datetime.datetime.now(datetime.timezone.utc).isoformat()
|
|
23
|
+
|
|
24
|
+
@dataclass
|
|
25
|
+
class ActiveConstraint:
|
|
26
|
+
code: str # e.g. "A_EVAL"
|
|
27
|
+
rule: str # human-readable enforced rule
|
|
28
|
+
promoted_from_count: int # how many observations triggered promotion
|
|
29
|
+
promoted_at: str
|
|
30
|
+
times_enforced: int = 0 # how often it has since blocked something
|
|
31
|
+
prevented_recurrence: int = 0 # times it caught the same failure again
|
|
32
|
+
status: str = "active" # active | probation | retired
|
|
33
|
+
|
|
34
|
+
class LearningKernel:
|
|
35
|
+
def __init__(self, root: Path, promote_threshold: int = 3):
|
|
36
|
+
self.root = Path(root)
|
|
37
|
+
self.threshold = promote_threshold
|
|
38
|
+
self.policy_path = self.root/POLICY
|
|
39
|
+
self.history_path = self.root/HISTORY
|
|
40
|
+
self.policy_path.parent.mkdir(parents=True, exist_ok=True)
|
|
41
|
+
|
|
42
|
+
def _load_policy(self) -> dict:
|
|
43
|
+
if self.policy_path.exists():
|
|
44
|
+
return json.loads(self.policy_path.read_text())
|
|
45
|
+
return {"constraints": {}, "version": 0}
|
|
46
|
+
|
|
47
|
+
def _save_policy(self, p: dict):
|
|
48
|
+
p["version"] = p.get("version", 0) + 1
|
|
49
|
+
p["updated"] = _now()
|
|
50
|
+
self.policy_path.write_text(json.dumps(p, indent=2))
|
|
51
|
+
|
|
52
|
+
def _log(self, event: str, **f):
|
|
53
|
+
f.update({"ts": _now(), "event": event})
|
|
54
|
+
with self.history_path.open("a") as fh:
|
|
55
|
+
fh.write(json.dumps(f, sort_keys=True) + "\n")
|
|
56
|
+
|
|
57
|
+
def promote(self, lessons: list[dict]) -> list[str]:
|
|
58
|
+
"""Tier 2: any lesson seen >= threshold becomes an active constraint.
|
|
59
|
+
Returns the codes newly promoted this cycle."""
|
|
60
|
+
policy = self._load_policy(); newly = []
|
|
61
|
+
for l in lessons:
|
|
62
|
+
code = l["failure_code"]
|
|
63
|
+
if l["count"] >= self.threshold and code not in policy["constraints"]:
|
|
64
|
+
c = ActiveConstraint(code=code, rule=l["fix"],
|
|
65
|
+
promoted_from_count=l["count"], promoted_at=_now())
|
|
66
|
+
policy["constraints"][code] = asdict(c)
|
|
67
|
+
newly.append(code)
|
|
68
|
+
self._log("promote", code=code, from_count=l["count"], rule=l["fix"])
|
|
69
|
+
if newly:
|
|
70
|
+
self._save_policy(policy)
|
|
71
|
+
return newly
|
|
72
|
+
|
|
73
|
+
def enforce(self, findings_codes: list[str]) -> dict:
|
|
74
|
+
"""Tier 3 measurement: record which active constraints fired this run.
|
|
75
|
+
A constraint that catches its target failure again = validated (working).
|
|
76
|
+
Returns {code: prevented_bool} for constraints that were relevant."""
|
|
77
|
+
policy = self._load_policy(); result = {}
|
|
78
|
+
for code in findings_codes:
|
|
79
|
+
if code in policy["constraints"]:
|
|
80
|
+
c = policy["constraints"][code]
|
|
81
|
+
c["times_enforced"] += 1
|
|
82
|
+
c["prevented_recurrence"] += 1
|
|
83
|
+
result[code] = True
|
|
84
|
+
self._log("enforce", code=code, prevented=True)
|
|
85
|
+
if result:
|
|
86
|
+
self._save_policy(policy)
|
|
87
|
+
return result
|
|
88
|
+
|
|
89
|
+
def audit_effectiveness(self, recent_run_count: int = 20) -> dict:
|
|
90
|
+
"""Recursive review: constraints that keep catching real failures are
|
|
91
|
+
VALIDATED; ones promoted long ago that never fire go to PROBATION
|
|
92
|
+
(candidate for retirement — the policy self-prunes)."""
|
|
93
|
+
policy = self._load_policy()
|
|
94
|
+
validated, probation = [], []
|
|
95
|
+
for code, c in policy["constraints"].items():
|
|
96
|
+
if c["status"] == "retired":
|
|
97
|
+
continue
|
|
98
|
+
if c["prevented_recurrence"] >= 1:
|
|
99
|
+
c["status"] = "active"; validated.append(code)
|
|
100
|
+
elif c["times_enforced"] == 0 and c["status"] == "active":
|
|
101
|
+
c["status"] = "probation"; probation.append(code)
|
|
102
|
+
self._log("probation", code=code, reason="never fired since promotion")
|
|
103
|
+
self._save_policy(policy)
|
|
104
|
+
return {"validated": validated, "probation": probation,
|
|
105
|
+
"total_active": sum(1 for c in policy["constraints"].values() if c["status"]=="active")}
|
|
106
|
+
|
|
107
|
+
def active_codes(self) -> list[str]:
|
|
108
|
+
policy = self._load_policy()
|
|
109
|
+
return [k for k,v in policy["constraints"].items() if v["status"] in ("active","probation")]
|
|
110
|
+
|
|
111
|
+
def policy_summary(self) -> dict:
|
|
112
|
+
policy = self._load_policy()
|
|
113
|
+
return {"version": policy.get("version",0),
|
|
114
|
+
"constraints": {k: {"status": v["status"], "enforced": v["times_enforced"],
|
|
115
|
+
"prevented": v["prevented_recurrence"]}
|
|
116
|
+
for k,v in policy["constraints"].items()}}
|
|
@@ -0,0 +1,263 @@
|
|
|
1
|
+
"""The outer loop. Advances the state machine, calling SpecLine for spec/plan
|
|
2
|
+
governance and HSF for decision compilation when a spec carries a decision
|
|
3
|
+
table. Everything is a receipt; every gate failure records a skill lesson and
|
|
4
|
+
routes to the refine loop."""
|
|
5
|
+
from __future__ import annotations
|
|
6
|
+
import shutil, subprocess, sys
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
from .states import State, can_transition, IllegalTransition, HUMAN_GATES
|
|
9
|
+
from .run_store import RunStore
|
|
10
|
+
from .ssat import load_ssat, scaffold_from_ssat, check_erosion
|
|
11
|
+
from .gates import judge_consistency, grumpy_review, skill_check
|
|
12
|
+
from .gates.qa_audit import qa_audit
|
|
13
|
+
from .skill_memory import record_lesson, inject_lessons_block, lessons_for
|
|
14
|
+
from .learning import LearningKernel
|
|
15
|
+
from .attribution import Attribution, FailureClass, UnitResult
|
|
16
|
+
|
|
17
|
+
MAX_REFINE = 3
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _gate_attribution(stage: str, checks: list[tuple[str, bool, str, FailureClass]]) -> dict:
|
|
21
|
+
units = [
|
|
22
|
+
UnitResult(
|
|
23
|
+
unit=f"{stage}:{name}",
|
|
24
|
+
stage=stage,
|
|
25
|
+
passed=passed,
|
|
26
|
+
evidence=evidence,
|
|
27
|
+
failure_class=None if passed else failure_class,
|
|
28
|
+
)
|
|
29
|
+
for name, passed, evidence, failure_class in checks
|
|
30
|
+
]
|
|
31
|
+
return Attribution(stage, len(units), sum(unit.passed for unit in units), units).to_dict()
|
|
32
|
+
|
|
33
|
+
class Orchestrator:
|
|
34
|
+
def __init__(self, root: Path, feature: str):
|
|
35
|
+
self.root = Path(root); self.feature = feature
|
|
36
|
+
self.store = RunStore(self.root, feature)
|
|
37
|
+
|
|
38
|
+
def _advance(self, to: State, note: str = "", **receipt):
|
|
39
|
+
frm = self.store.state
|
|
40
|
+
if not can_transition(frm, to):
|
|
41
|
+
raise IllegalTransition(f"{frm.value} -> {to.value}")
|
|
42
|
+
self.store.set_state(to, note)
|
|
43
|
+
self.store.receipt(transition=f"{frm.value}->{to.value}", **receipt)
|
|
44
|
+
|
|
45
|
+
def architect(self, ssat_path: Path) -> dict:
|
|
46
|
+
ssat = load_ssat(ssat_path)
|
|
47
|
+
created = scaffold_from_ssat(ssat, self.root)
|
|
48
|
+
self._advance(State.SCAFFOLDED, "scaffold from SSAT",
|
|
49
|
+
modules=len(ssat.get("modules", [])), files=[str(c) for c in created])
|
|
50
|
+
return {"scaffolded": [str(c) for c in created]}
|
|
51
|
+
|
|
52
|
+
def review(self, ssat_path: Path) -> dict:
|
|
53
|
+
"""Judge + grumpy adversary + arch erosion + deep QA audit, with a
|
|
54
|
+
recursive learning kernel and escalating refine loop."""
|
|
55
|
+
ssat = load_ssat(ssat_path)
|
|
56
|
+
attempt = self.store.bump_attempt("review")
|
|
57
|
+
kernel = LearningKernel(self.root)
|
|
58
|
+
j_ok, j = judge_consistency(ssat, self.root)
|
|
59
|
+
g_ok, g = grumpy_review(self.root)
|
|
60
|
+
erosion = check_erosion(ssat, self.root)
|
|
61
|
+
qa = qa_audit(self.root) # STRICTER: quantitative QA grade
|
|
62
|
+
all_findings = j + g + [f"{v.code} {v.message}" for v in erosion] + qa.findings
|
|
63
|
+
all_ok = j_ok and g_ok and not erosion and qa.passed
|
|
64
|
+
review_attr = _gate_attribution("review", [
|
|
65
|
+
("judge", j_ok, "consistent" if j_ok else " | ".join(j),
|
|
66
|
+
FailureClass.INCONSISTENT_LOGIC),
|
|
67
|
+
("adversary", g_ok, "all probes resisted" if g_ok else " | ".join(g),
|
|
68
|
+
FailureClass.SECURITY_FINDING),
|
|
69
|
+
("architecture", not erosion, "no erosion" if not erosion else
|
|
70
|
+
" | ".join(f"{v.code} {v.message}" for v in erosion),
|
|
71
|
+
FailureClass.SIGNATURE_DRIFT),
|
|
72
|
+
("qa_audit", qa.passed, f"grade={qa.grade}; metrics={qa.metrics}",
|
|
73
|
+
FailureClass.COMPLEXITY_EXCEEDED),
|
|
74
|
+
])
|
|
75
|
+
|
|
76
|
+
# recursive learning: measure whether active constraints caught failures
|
|
77
|
+
codes = [f.split()[0].split("[")[0] for f in all_findings]
|
|
78
|
+
prevented = kernel.enforce(codes)
|
|
79
|
+
|
|
80
|
+
self.store.receipt(phase="review", attempt=attempt, judge_ok=j_ok,
|
|
81
|
+
grumpy_ok=g_ok, erosion=len(erosion),
|
|
82
|
+
qa_grade=qa.grade, qa_metrics=qa.metrics,
|
|
83
|
+
active_constraints_fired=list(prevented.keys()),
|
|
84
|
+
findings=all_findings, attribution=review_attr)
|
|
85
|
+
if all_ok:
|
|
86
|
+
self._advance(State.REVIEWED, f"review pass (attempt {attempt}, QA={qa.grade})")
|
|
87
|
+
return {"reviewed": True, "attempt": attempt, "qa_grade": qa.grade,
|
|
88
|
+
"qa_metrics": qa.metrics, "attribution": review_attr}
|
|
89
|
+
|
|
90
|
+
# record lessons, then PROMOTE recurring ones into active policy (recursion)
|
|
91
|
+
for f in all_findings:
|
|
92
|
+
code = f.split()[0].split("[")[0]
|
|
93
|
+
record_lesson(self.root, phase="fill", failure_code=code, fix=f, feature=self.feature)
|
|
94
|
+
promoted = kernel.promote(lessons_for(self.root, "fill"))
|
|
95
|
+
kernel.audit_effectiveness()
|
|
96
|
+
|
|
97
|
+
if self.store.state != State.BLOCKED:
|
|
98
|
+
self._advance(State.BLOCKED, f"review failed (attempt {attempt}, QA={qa.grade})")
|
|
99
|
+
|
|
100
|
+
# ESCALATION: refine loop gets stricter each attempt
|
|
101
|
+
escalation = ("normal" if attempt == 1 else
|
|
102
|
+
"elevated: fix ALL findings, not just blockers" if attempt == 2 else
|
|
103
|
+
"final: exhausted — human review required" )
|
|
104
|
+
result = {"reviewed": False, "attempt": attempt, "qa_grade": qa.grade,
|
|
105
|
+
"qa_metrics": qa.metrics, "findings": all_findings,
|
|
106
|
+
"newly_promoted_constraints": promoted, "escalation": escalation,
|
|
107
|
+
"lessons_for_next": inject_lessons_block(self.root, "fill"),
|
|
108
|
+
"attribution": review_attr}
|
|
109
|
+
if attempt >= MAX_REFINE:
|
|
110
|
+
result["exhausted"] = True
|
|
111
|
+
return result
|
|
112
|
+
|
|
113
|
+
def arch_gate(self, ssat_path: Path) -> dict:
|
|
114
|
+
ssat = load_ssat(ssat_path)
|
|
115
|
+
erosion = check_erosion(ssat, self.root)
|
|
116
|
+
s_ok, s = skill_check(self.root, self.feature)
|
|
117
|
+
arch_attr = _gate_attribution("arch_gate", [
|
|
118
|
+
("erosion", not erosion, "no architecture erosion" if not erosion else
|
|
119
|
+
" | ".join(f"{v.code} {v.message}" for v in erosion),
|
|
120
|
+
FailureClass.SIGNATURE_DRIFT),
|
|
121
|
+
("skill_contract", s_ok, "skill contract valid" if s_ok else " | ".join(s),
|
|
122
|
+
FailureClass.INCONSISTENT_LOGIC),
|
|
123
|
+
])
|
|
124
|
+
if erosion or not s_ok:
|
|
125
|
+
self.store.receipt(phase="arch_gate", passed=False,
|
|
126
|
+
erosion=[f"{v.code} {v.message}" for v in erosion], skill=s,
|
|
127
|
+
attribution=arch_attr)
|
|
128
|
+
if self.store.state != State.BLOCKED:
|
|
129
|
+
self._advance(State.BLOCKED, "arch gate failed")
|
|
130
|
+
return {"passed": False, "erosion": [f"{v.code} {v.message}" for v in erosion] + s,
|
|
131
|
+
"attribution": arch_attr}
|
|
132
|
+
# must currently be REVIEWED to legally gate
|
|
133
|
+
if self.store.state == State.BLOCKED:
|
|
134
|
+
self.store.set_state(State.REVIEWED, "recovered for arch gate")
|
|
135
|
+
kernel = LearningKernel(self.root)
|
|
136
|
+
eff = kernel.audit_effectiveness()
|
|
137
|
+
self._advance(State.ARCH_GATED, "architecture CI gate passed")
|
|
138
|
+
self.store.receipt(phase="arch_gate", passed=True, learning=eff,
|
|
139
|
+
active_policy=kernel.policy_summary(), attribution=arch_attr)
|
|
140
|
+
return {"passed": True, "learning": eff, "attribution": arch_attr}
|
|
141
|
+
|
|
142
|
+
def handoff_decisions(self, spec_path: Path) -> dict | None:
|
|
143
|
+
"""If SpecLine + HSF are importable and the spec has a decision table,
|
|
144
|
+
compile decisions through the factory. Best-effort seam."""
|
|
145
|
+
try:
|
|
146
|
+
sys.path.insert(0, str(Path(__file__).resolve().parents[2]/"specline"))
|
|
147
|
+
from specline.spec_lint import decision_rows # type: ignore
|
|
148
|
+
except Exception:
|
|
149
|
+
return None
|
|
150
|
+
rows = decision_rows(spec_path)
|
|
151
|
+
if not rows:
|
|
152
|
+
return None
|
|
153
|
+
self.store.receipt(phase="handoff", decision_rules=len(rows),
|
|
154
|
+
note="decision table present — route to HSF compile")
|
|
155
|
+
return {"decision_rules": len(rows), "next": "hsf compile"}
|
|
156
|
+
|
|
157
|
+
def smoke_gate(self) -> dict:
|
|
158
|
+
"""Runtime behavior verification — the behavior-by-inspection gate.
|
|
159
|
+
Runs the built artifact against declared behavioral checks (smoke/<feature>.json)
|
|
160
|
+
in isolated subprocesses. Blocks ship on any runtime failure. This is the
|
|
161
|
+
right-sized version of a per-PR preview deployment: verify the thing RUNS
|
|
162
|
+
and behaves, not just that it type-checks and honors the spec."""
|
|
163
|
+
from .gates.runtime_smoke import runtime_smoke, smoke_report_lines
|
|
164
|
+
if self.store.state == State.SHIPPED:
|
|
165
|
+
return {"smoked": True, "note": "already shipped"}
|
|
166
|
+
if self.store.state == State.SMOKED:
|
|
167
|
+
return {"smoked": True, "note": "already smoked"}
|
|
168
|
+
if self.store.state != State.TESTS_VERIFIED:
|
|
169
|
+
return {"smoked": False,
|
|
170
|
+
"reason": "tests not verified - run `forge verify-tests` first"}
|
|
171
|
+
rep = runtime_smoke(self.root, self.feature)
|
|
172
|
+
gate = rep.gate_result
|
|
173
|
+
attr = gate.attribution.to_dict()
|
|
174
|
+
lines = smoke_report_lines(rep)
|
|
175
|
+
if not gate.passed:
|
|
176
|
+
self.store.receipt(phase="smoke_gate", smoked=False,
|
|
177
|
+
manifest_found=rep.manifest_found,
|
|
178
|
+
failures=[r.name for r in rep.failures],
|
|
179
|
+
report=lines, attribution=attr)
|
|
180
|
+
# a runtime failure is a real defect -> block; refine loop owns recovery
|
|
181
|
+
self._advance(State.BLOCKED, "runtime smoke gate failed")
|
|
182
|
+
return {"smoked": False,
|
|
183
|
+
"reason": ("no runtime behavior verified" if not rep.manifest_found
|
|
184
|
+
else "runtime behavioral check(s) failed"),
|
|
185
|
+
"failures": [f"{r.name}: {r.reason}" for r in rep.failures],
|
|
186
|
+
"attribution": attr}
|
|
187
|
+
self._advance(State.SMOKED, "runtime behavior verified")
|
|
188
|
+
self.store.receipt(phase="smoke_gate", smoked=True,
|
|
189
|
+
checks=len(rep.results), report=lines, attribution=attr)
|
|
190
|
+
return {"smoked": True, "checks": len(rep.results), "attribution": attr}
|
|
191
|
+
|
|
192
|
+
def verify_tests(self, ssat_path: Path) -> dict:
|
|
193
|
+
"""Verify smoke checks can fail before trusting smoke results."""
|
|
194
|
+
from .gates.reverse_classical import verify_tests
|
|
195
|
+
if self.store.state == State.SHIPPED:
|
|
196
|
+
return {"verified": True, "note": "already shipped"}
|
|
197
|
+
if self.store.state in {State.TESTS_VERIFIED, State.SMOKED}:
|
|
198
|
+
return {"verified": True, "note": "already verified"}
|
|
199
|
+
if self.store.state not in {State.ARCH_GATED, State.BLOCKED}:
|
|
200
|
+
return {"verified": False,
|
|
201
|
+
"reason": "architecture gate not passed - run `forge arch-gate` first"}
|
|
202
|
+
gate = verify_tests(self.root, self.feature, Path(ssat_path))
|
|
203
|
+
attr = gate.attribution.to_dict()
|
|
204
|
+
if not gate.passed:
|
|
205
|
+
self.store.receipt(phase="verify_tests", verified=False, attribution=attr)
|
|
206
|
+
if self.store.state != State.BLOCKED:
|
|
207
|
+
self._advance(State.BLOCKED, "reverse-classical test verification failed")
|
|
208
|
+
return {"verified": False,
|
|
209
|
+
"reason": "hollow smoke check(s) found",
|
|
210
|
+
"attribution": attr}
|
|
211
|
+
self._advance(State.TESTS_VERIFIED, "smoke checks proven non-hollow",
|
|
212
|
+
attribution=attr)
|
|
213
|
+
self.store.receipt(phase="verify_tests", verified=True, attribution=attr)
|
|
214
|
+
return {"verified": True, "checks": gate.attribution.n_checked, "attribution": attr}
|
|
215
|
+
|
|
216
|
+
def refine(self, evaluate, propose, apply, revert, max_iters: int = 6) -> dict:
|
|
217
|
+
"""Run deterministic localized refinement.
|
|
218
|
+
|
|
219
|
+
Callers may propose an edit, but exact Pareto acceptance and rollback are
|
|
220
|
+
controlled here. Learning state remains build-time only under `.forge/`.
|
|
221
|
+
"""
|
|
222
|
+
from .refinement import refine
|
|
223
|
+
return refine(evaluate, propose, apply, revert, self.root, max_iters=max_iters)
|
|
224
|
+
|
|
225
|
+
def ship(self, verify_intent: bool = True) -> dict:
|
|
226
|
+
# ship now requires runtime behavior to have been verified
|
|
227
|
+
if self.store.state in {State.ARCH_GATED, State.TESTS_VERIFIED}:
|
|
228
|
+
return {"shipped": False, "reason": "runtime smoke gate not run - "
|
|
229
|
+
"call smoke_gate() before ship (behavior must be verified)."}
|
|
230
|
+
trace = None
|
|
231
|
+
intent_attr = None
|
|
232
|
+
if verify_intent:
|
|
233
|
+
from .intent_thread import verify_against_intent
|
|
234
|
+
trace = verify_against_intent(self.root, self.feature, self.root)
|
|
235
|
+
intent_checks = [
|
|
236
|
+
(f"obligation_{index}", met, evidence, FailureClass.INCONSISTENT_LOGIC)
|
|
237
|
+
for index, (met, evidence) in enumerate(
|
|
238
|
+
[(not trace.unverified_assumptions, finding)
|
|
239
|
+
for finding in (trace.findings or ["intent obligations satisfied"])],
|
|
240
|
+
1,
|
|
241
|
+
)
|
|
242
|
+
]
|
|
243
|
+
intent_attr = _gate_attribution("intent_thread", intent_checks)
|
|
244
|
+
if trace.envelope_found and not trace.traceable:
|
|
245
|
+
# PRD->production gap: shipped code doesn't honor sealed intent
|
|
246
|
+
self.store.receipt(phase="ship", shipped=False,
|
|
247
|
+
intent_hash=trace.intent_hash,
|
|
248
|
+
unverified_assumptions=trace.unverified_assumptions,
|
|
249
|
+
findings=trace.findings, attribution=intent_attr)
|
|
250
|
+
if self.store.state != State.BLOCKED:
|
|
251
|
+
self._advance(State.BLOCKED, "intent-traceability gap")
|
|
252
|
+
return {"shipped": False, "reason": "intent not honored by code",
|
|
253
|
+
"unverified_assumptions": trace.unverified_assumptions,
|
|
254
|
+
"findings": trace.findings}
|
|
255
|
+
self._advance(State.SHIPPED, "all gates green")
|
|
256
|
+
self.store.receipt(phase="ship", shipped=True,
|
|
257
|
+
intent_hash=(trace.intent_hash if trace else None),
|
|
258
|
+
obligations=f"{trace.obligations_met}/{trace.obligations_total}" if trace else None,
|
|
259
|
+
attribution=intent_attr)
|
|
260
|
+
return {"shipped": True,
|
|
261
|
+
"intent_traceable": (trace.traceable if trace else None),
|
|
262
|
+
"obligations_met": (f"{trace.obligations_met}/{trace.obligations_total}" if trace else None),
|
|
263
|
+
"attribution": intent_attr}
|
forgeline/refinement.py
ADDED
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
"""Deterministic refinement policy: one edit, ordered, Pareto-gated, bounded."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
from dataclasses import asdict, dataclass
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
import json
|
|
6
|
+
|
|
7
|
+
from .attribution import FailureClass
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
@dataclass(frozen=True)
|
|
11
|
+
class Edit:
|
|
12
|
+
edit_class: str
|
|
13
|
+
target: str
|
|
14
|
+
description: str
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
STRUCTURAL_FAILURES = {
|
|
18
|
+
FailureClass.SIGNATURE_DRIFT,
|
|
19
|
+
FailureClass.STUB_UNFILLED,
|
|
20
|
+
FailureClass.INCONSISTENT_LOGIC,
|
|
21
|
+
FailureClass.SCOPE_ESCAPE,
|
|
22
|
+
FailureClass.HOLLOW_TEST,
|
|
23
|
+
FailureClass.HOLLOW_MANIFEST,
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def select_edit(failure_class: FailureClass) -> Edit:
|
|
28
|
+
edit_class = "structural" if failure_class in STRUCTURAL_FAILURES else (
|
|
29
|
+
"configuration" if failure_class in {
|
|
30
|
+
FailureClass.RUNTIME_CRASH, FailureClass.RUNTIME_TIMEOUT,
|
|
31
|
+
FailureClass.SECURITY_FINDING,
|
|
32
|
+
} else "parametric"
|
|
33
|
+
)
|
|
34
|
+
return Edit(edit_class, failure_class.value, f"localized {edit_class} correction")
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def pareto_win(current: dict[str, float], previous: dict[str, float], target: str) -> bool:
|
|
38
|
+
return (
|
|
39
|
+
current.get(target, 0.0) > previous.get(target, 0.0)
|
|
40
|
+
and all(current.get(stage, 0.0) >= rate for stage, rate in previous.items())
|
|
41
|
+
)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
class RejectionLedger:
|
|
45
|
+
def __init__(self, root: Path):
|
|
46
|
+
self.path = Path(root) / ".forge" / "rejection_ledger.jsonl"
|
|
47
|
+
|
|
48
|
+
def log(self, edit: Edit, before: dict, after: dict) -> None:
|
|
49
|
+
self.path.parent.mkdir(parents=True, exist_ok=True)
|
|
50
|
+
with self.path.open("a", encoding="utf-8") as stream:
|
|
51
|
+
stream.write(json.dumps({
|
|
52
|
+
"edit": asdict(edit), "before_rates": before, "after_rates": after,
|
|
53
|
+
}, sort_keys=True) + "\n")
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def refine(evaluate, propose, apply, revert, root: Path, max_iters: int = 6) -> dict:
|
|
57
|
+
previous = evaluate()
|
|
58
|
+
no_win_streak = 0
|
|
59
|
+
ledger = RejectionLedger(root)
|
|
60
|
+
for iteration in range(max_iters):
|
|
61
|
+
if all(rate == 1.0 for rate in previous.values()):
|
|
62
|
+
return {"converged": True, "iters": iteration}
|
|
63
|
+
target, failure_class = propose(previous)
|
|
64
|
+
edit = select_edit(failure_class)
|
|
65
|
+
snapshot = apply(edit)
|
|
66
|
+
current = evaluate()
|
|
67
|
+
if pareto_win(current, previous, target):
|
|
68
|
+
previous = current
|
|
69
|
+
no_win_streak = 0
|
|
70
|
+
continue
|
|
71
|
+
revert(snapshot)
|
|
72
|
+
ledger.log(edit, previous, current)
|
|
73
|
+
no_win_streak += 1
|
|
74
|
+
if no_win_streak >= 2:
|
|
75
|
+
return {"converged": False, "reason": "plateau", "iters": iteration + 1}
|
|
76
|
+
return {"converged": False, "reason": "budget", "iters": max_iters}
|
forgeline/run_store.py
ADDED
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
"""Per-feature run state + receipts. The state machine's durable memory —
|
|
2
|
+
survives context resets (Ralph Wiggum) because the disk is the truth."""
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
import json, datetime, hashlib
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from .states import State
|
|
7
|
+
|
|
8
|
+
def _now():
|
|
9
|
+
return datetime.datetime.now(datetime.timezone.utc).isoformat()
|
|
10
|
+
|
|
11
|
+
class RunStore:
|
|
12
|
+
def __init__(self, root: Path, feature: str):
|
|
13
|
+
self.root = Path(root); self.feature = feature
|
|
14
|
+
self.dir = self.root/".forge"/feature
|
|
15
|
+
self.dir.mkdir(parents=True, exist_ok=True)
|
|
16
|
+
self.state_path = self.dir/"state.json"
|
|
17
|
+
self.receipts = self.dir/"receipts.jsonl"
|
|
18
|
+
if not self.state_path.exists():
|
|
19
|
+
self._write({"feature": feature, "state": State.INTENT.value,
|
|
20
|
+
"created": _now(), "attempts": {}, "history": []})
|
|
21
|
+
|
|
22
|
+
def _write(self, data): self.state_path.write_text(json.dumps(data, indent=2))
|
|
23
|
+
def load(self) -> dict: return json.loads(self.state_path.read_text())
|
|
24
|
+
|
|
25
|
+
@property
|
|
26
|
+
def state(self) -> State:
|
|
27
|
+
return State(self.load()["state"])
|
|
28
|
+
|
|
29
|
+
def set_state(self, s: State, note: str = ""):
|
|
30
|
+
d = self.load(); d["state"] = s.value
|
|
31
|
+
d["history"].append({"ts": _now(), "state": s.value, "note": note})
|
|
32
|
+
self._write(d)
|
|
33
|
+
|
|
34
|
+
def bump_attempt(self, phase: str) -> int:
|
|
35
|
+
d = self.load(); d["attempts"][phase] = d["attempts"].get(phase, 0) + 1
|
|
36
|
+
self._write(d); return d["attempts"][phase]
|
|
37
|
+
|
|
38
|
+
def receipt(self, **fields):
|
|
39
|
+
fields["ts"] = _now()
|
|
40
|
+
line = json.dumps(fields, sort_keys=True)
|
|
41
|
+
fields = {"h": hashlib.sha256(line.encode()).hexdigest()[:12], **fields}
|
|
42
|
+
with self.receipts.open("a") as f:
|
|
43
|
+
f.write(json.dumps(fields, sort_keys=True) + "\n")
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
"""Skill memory — the self-accelerating flywheel. Records what failed in each
|
|
2
|
+
run and the fix, so future runs inject accumulated lessons into agent context.
|
|
3
|
+
This is the 'skill learns from past refinements' mechanism, made concrete and
|
|
4
|
+
LLM-free: lessons are structured, deduped, and promotable to constraints."""
|
|
5
|
+
from __future__ import annotations
|
|
6
|
+
import json, datetime
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
LESSONS = "skills/lessons.jsonl"
|
|
10
|
+
|
|
11
|
+
def record_lesson(root: Path, *, phase: str, failure_code: str, fix: str, feature: str):
|
|
12
|
+
root = Path(root); lp = root/LESSONS
|
|
13
|
+
lp.parent.mkdir(parents=True, exist_ok=True)
|
|
14
|
+
# dedupe by (phase, failure_code, fix)
|
|
15
|
+
existing = []
|
|
16
|
+
if lp.exists():
|
|
17
|
+
existing = [json.loads(l) for l in lp.read_text().splitlines() if l.strip()]
|
|
18
|
+
key = (phase, failure_code, fix)
|
|
19
|
+
for e in existing:
|
|
20
|
+
if (e["phase"], e["failure_code"], e["fix"]) == key:
|
|
21
|
+
e["count"] += 1; e["last_seen"] = datetime.datetime.now(datetime.timezone.utc).isoformat()
|
|
22
|
+
lp.write_text("\n".join(json.dumps(x) for x in existing) + "\n")
|
|
23
|
+
return e
|
|
24
|
+
entry = {"phase": phase, "failure_code": failure_code, "fix": fix, "feature": feature,
|
|
25
|
+
"count": 1, "last_seen": datetime.datetime.now(datetime.timezone.utc).isoformat()}
|
|
26
|
+
existing.append(entry)
|
|
27
|
+
lp.write_text("\n".join(json.dumps(x) for x in existing) + "\n")
|
|
28
|
+
return entry
|
|
29
|
+
|
|
30
|
+
def lessons_for(root: Path, phase: str, min_count: int = 1) -> list[dict]:
|
|
31
|
+
lp = Path(root)/LESSONS
|
|
32
|
+
if not lp.exists(): return []
|
|
33
|
+
rows = [json.loads(l) for l in lp.read_text().splitlines() if l.strip()]
|
|
34
|
+
return sorted([r for r in rows if r["phase"] == phase and r["count"] >= min_count],
|
|
35
|
+
key=lambda r: -r["count"])
|
|
36
|
+
|
|
37
|
+
def promotable_constraints(root: Path, threshold: int = 3) -> list[dict]:
|
|
38
|
+
"""Lessons seen >= threshold times graduate into hard constraints
|
|
39
|
+
(conventions-into-constraints). These get injected as SSAT invariants."""
|
|
40
|
+
lp = Path(root)/LESSONS
|
|
41
|
+
if not lp.exists(): return []
|
|
42
|
+
rows = [json.loads(l) for l in lp.read_text().splitlines() if l.strip()]
|
|
43
|
+
return [r for r in rows if r["count"] >= threshold]
|
|
44
|
+
|
|
45
|
+
def inject_lessons_block(root: Path, phase: str) -> str:
|
|
46
|
+
"""Text block injected into an agent's task context for this phase."""
|
|
47
|
+
ls = lessons_for(root, phase)
|
|
48
|
+
if not ls: return ""
|
|
49
|
+
lines = [f"## Lessons from past runs (phase: {phase})"]
|
|
50
|
+
for l in ls[:8]:
|
|
51
|
+
lines.append(f"- [{l['failure_code']} ×{l['count']}] {l['fix']}")
|
|
52
|
+
return "\n".join(lines)
|