code-factory-2-forge 0.6.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- code_factory_2_forge-0.6.0.dist-info/METADATA +308 -0
- code_factory_2_forge-0.6.0.dist-info/RECORD +29 -0
- code_factory_2_forge-0.6.0.dist-info/WHEEL +5 -0
- code_factory_2_forge-0.6.0.dist-info/entry_points.txt +2 -0
- code_factory_2_forge-0.6.0.dist-info/licenses/LICENSE-APACHE +18 -0
- code_factory_2_forge-0.6.0.dist-info/licenses/LICENSE-MIT +21 -0
- code_factory_2_forge-0.6.0.dist-info/licenses/NOTICE +6 -0
- code_factory_2_forge-0.6.0.dist-info/top_level.txt +1 -0
- forgeline/__init__.py +14 -0
- forgeline/adapters.py +56 -0
- forgeline/attribution.py +71 -0
- forgeline/cli.py +101 -0
- forgeline/demo.py +83 -0
- forgeline/demo_learning.py +44 -0
- forgeline/gates/__init__.py +4 -0
- forgeline/gates/adversary.py +45 -0
- forgeline/gates/judge.py +26 -0
- forgeline/gates/qa_audit.py +134 -0
- forgeline/gates/reverse_classical.py +118 -0
- forgeline/gates/runtime_smoke.py +194 -0
- forgeline/gates/skill_check.py +14 -0
- forgeline/intent_thread.py +76 -0
- forgeline/learning.py +116 -0
- forgeline/orchestrator.py +263 -0
- forgeline/refinement.py +76 -0
- forgeline/run_store.py +43 -0
- forgeline/skill_memory.py +52 -0
- forgeline/ssat.py +93 -0
- forgeline/states.py +41 -0
forgeline/demo.py
ADDED
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
"""forge demo - a headless walk through refine loop + skill flywheel."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import shutil
|
|
5
|
+
import tempfile
|
|
6
|
+
import time
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
C = {"g": "\033[92m", "c": "\033[96m", "y": "\033[93m", "d": "\033[2m", "b": "\033[1m", "x": "\033[0m"}
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def p(message: str, sleep: float = 0.02) -> None:
|
|
13
|
+
print(message.encode("ascii", "replace").decode("ascii"))
|
|
14
|
+
time.sleep(sleep)
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def run() -> None:
|
|
18
|
+
from forgeline.orchestrator import Orchestrator
|
|
19
|
+
from forgeline.skill_memory import lessons_for
|
|
20
|
+
from forgeline.states import State
|
|
21
|
+
|
|
22
|
+
root = Path(tempfile.mkdtemp())
|
|
23
|
+
(root / "skills").mkdir()
|
|
24
|
+
ssat_src = Path(__file__).resolve().parents[1] / "examples" / "notifier.ssat.yaml"
|
|
25
|
+
ssat = root / "notifier.ssat.yaml"
|
|
26
|
+
shutil.copy(ssat_src, ssat)
|
|
27
|
+
|
|
28
|
+
p(f"\n{C['b']}* ForgeLine - autonomous factory outer loop{C['x']}")
|
|
29
|
+
p(f"{C['d']} intent -> SSAT -> scaffold -> fill -> adversarial review -> ship{C['x']}\n")
|
|
30
|
+
|
|
31
|
+
orchestrator = Orchestrator(root, "notifier")
|
|
32
|
+
orchestrator.store.set_state(State.ARCHITECTED)
|
|
33
|
+
p(f"{C['c']}[1] architect{C['x']} - scaffolding signatures from architecture-as-code")
|
|
34
|
+
orchestrator.architect(ssat)
|
|
35
|
+
p(f" {C['g']}OK{C['x']} 2 modules scaffolded with valid imports\n")
|
|
36
|
+
|
|
37
|
+
slices = root / "slices" / "notifier"
|
|
38
|
+
(slices / "formatter.py").write_text(
|
|
39
|
+
'def format_message(event: dict) -> str:\n return str(eval(event.get("expr","1")))\n',
|
|
40
|
+
encoding="utf-8",
|
|
41
|
+
)
|
|
42
|
+
(slices / "sender.py").write_text(
|
|
43
|
+
"from slices.notifier.formatter import format_message\n"
|
|
44
|
+
"def send(event: dict, channel: str) -> bool:\n"
|
|
45
|
+
" return bool(format_message(event) and channel)\n",
|
|
46
|
+
encoding="utf-8",
|
|
47
|
+
)
|
|
48
|
+
orchestrator.store.set_state(State.FILLED)
|
|
49
|
+
p(f"{C['c']}[2] fill (attempt 1){C['x']} - agent used eval(), shipped no tests")
|
|
50
|
+
review = orchestrator.review(ssat)
|
|
51
|
+
p(f" {C['y']}BLOCKED: grumpy adversary findings:{C['x']}")
|
|
52
|
+
for finding in review["findings"]:
|
|
53
|
+
p(f" - {finding}")
|
|
54
|
+
p(f" {C['d']}lesson recorded to skill memory{C['x']}\n")
|
|
55
|
+
|
|
56
|
+
p(f"{C['c']}[3] lessons injected into next context:{C['x']}")
|
|
57
|
+
for lesson in lessons_for(root, "fill"):
|
|
58
|
+
p(f" [{lesson['failure_code']} x{lesson['count']}] {lesson['fix'][:60]}")
|
|
59
|
+
|
|
60
|
+
(slices / "formatter.py").write_text(
|
|
61
|
+
"def format_message(event: dict) -> str:\n"
|
|
62
|
+
" return f\"{event.get('kind','x')}: {event.get('text','')}\"\n",
|
|
63
|
+
encoding="utf-8",
|
|
64
|
+
)
|
|
65
|
+
(root / "tests").mkdir(exist_ok=True)
|
|
66
|
+
(root / "tests" / "test_notifier.py").write_text(
|
|
67
|
+
"def test_ok():\n assert True\n",
|
|
68
|
+
encoding="utf-8",
|
|
69
|
+
)
|
|
70
|
+
p(f"\n{C['c']}[4] fill (attempt 2){C['x']} - eval removed, tests supplied")
|
|
71
|
+
review2 = orchestrator.review(ssat)
|
|
72
|
+
if review2["reviewed"]:
|
|
73
|
+
p(f" {C['g']}OK judge + grumpy + arch erosion all pass{C['x']}")
|
|
74
|
+
orchestrator.arch_gate(ssat)
|
|
75
|
+
orchestrator.ship()
|
|
76
|
+
p(f" {C['g']}OK architecture CI gate pass -> SHIPPED{C['x']}\n")
|
|
77
|
+
|
|
78
|
+
p(f"{C['d']}The factory caught its own bad output, learned, and shipped clean.")
|
|
79
|
+
p(f" Wire it in: forge agent claude | forge agent codex{C['x']}\n")
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
if __name__ == "__main__":
|
|
83
|
+
run()
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
"""forge demo-learning — shows the recursive loop: a failure recurs, gets
|
|
2
|
+
promoted into active policy, then that policy catches it on the next run and
|
|
3
|
+
is marked validated. The factory improving its own QA."""
|
|
4
|
+
from __future__ import annotations
|
|
5
|
+
import tempfile, time
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from .learning import LearningKernel
|
|
8
|
+
from .skill_memory import record_lesson, lessons_for
|
|
9
|
+
|
|
10
|
+
C = {"g":"\033[92m","c":"\033[96m","y":"\033[93m","d":"\033[2m","b":"\033[1m","x":"\033[0m"}
|
|
11
|
+
def p(m): print(m); time.sleep(0.02)
|
|
12
|
+
|
|
13
|
+
def run():
|
|
14
|
+
root = Path(tempfile.mkdtemp()); (root/"skills").mkdir()
|
|
15
|
+
k = LearningKernel(root, promote_threshold=3)
|
|
16
|
+
p(f"\n{C['b']}✦ ForgeLine — recursive learning loop{C['x']}")
|
|
17
|
+
p(f"{C['d']} observe → promote → validate → self-prune{C['x']}\n")
|
|
18
|
+
|
|
19
|
+
p(f"{C['c']}[1] observe{C['x']} — the same flaw (eval) appears in 3 runs")
|
|
20
|
+
for i in range(3):
|
|
21
|
+
record_lesson(root, phase="fill", failure_code="A_EVAL", fix="never use eval()", feature=f"run{i}")
|
|
22
|
+
p(f" {C['y']}▲{C['x']} run {i+1}: A_EVAL recorded")
|
|
23
|
+
|
|
24
|
+
p(f"\n{C['c']}[2] promote{C['x']} — seen ≥3× → becomes an ENFORCED constraint")
|
|
25
|
+
promoted = k.promote(lessons_for(root, "fill"))
|
|
26
|
+
p(f" {C['g']}✓{C['x']} promoted to active policy: {promoted}")
|
|
27
|
+
|
|
28
|
+
p(f"\n{C['c']}[3] validate{C['x']} — next run, the policy CATCHES the same flaw")
|
|
29
|
+
prevented = k.enforce(["A_EVAL"])
|
|
30
|
+
p(f" {C['g']}✓{C['x']} A_EVAL caught by learned policy → validated")
|
|
31
|
+
|
|
32
|
+
p(f"\n{C['c']}[4] self-prune{C['x']} — a stale rule that never fires goes to probation")
|
|
33
|
+
for i in range(3):
|
|
34
|
+
record_lesson(root, phase="fill", failure_code="A_OLD", fix="obsolete rule", feature="x")
|
|
35
|
+
k.promote(lessons_for(root, "fill"))
|
|
36
|
+
eff = k.audit_effectiveness()
|
|
37
|
+
p(f" {C['d']}validated: {eff['validated']} · probation: {eff['probation']}{C['x']}")
|
|
38
|
+
|
|
39
|
+
s = k.policy_summary()
|
|
40
|
+
p(f"\n{C['d']}Active policy v{s['version']} — the factory now enforces what it learned.")
|
|
41
|
+
p(f" Every future build is checked against its own accumulated experience.{C['x']}\n")
|
|
42
|
+
|
|
43
|
+
if __name__ == "__main__":
|
|
44
|
+
run()
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
"""The grumpy adversary — assumes the code is broken/insecure and makes the
|
|
2
|
+
generator prove otherwise. Executable heuristics (no LLM needed to be
|
|
3
|
+
useful); an optional LLM adversary can be layered behind the same interface.
|
|
4
|
+
It does NOT need to be right — it needs to force proof (tests + arch completeness)."""
|
|
5
|
+
from __future__ import annotations
|
|
6
|
+
import ast, re
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
DANGER = {
|
|
10
|
+
"eval(": "A_EVAL arbitrary eval",
|
|
11
|
+
"exec(": "A_EXEC arbitrary exec",
|
|
12
|
+
"os.system": "A_SHELL shell execution",
|
|
13
|
+
"subprocess": "A_SUBPROC subprocess use",
|
|
14
|
+
"pickle.load": "A_PICKLE unsafe deserialization",
|
|
15
|
+
"verify=False": "A_TLS TLS verification disabled",
|
|
16
|
+
"shell=True": "A_SHELL_TRUE shell=True injection surface",
|
|
17
|
+
}
|
|
18
|
+
SECRET = re.compile(r"(?i)(api[_-]?key|secret|password|token)\s*=\s*['\"][A-Za-z0-9/\+_-]{12,}['\"]")
|
|
19
|
+
|
|
20
|
+
def grumpy_review(src_dir: Path, require_tests: bool = True) -> tuple[bool, list[str]]:
|
|
21
|
+
"""Returns (satisfied, complaints). Grumpy is satisfied only when it can
|
|
22
|
+
find NO obvious sin AND the generator supplied tests to prove correctness."""
|
|
23
|
+
src_dir = Path(src_dir); complaints = []
|
|
24
|
+
py = list(src_dir.rglob("*.py"))
|
|
25
|
+
for p in py:
|
|
26
|
+
if p.name.startswith("test_") or p.parent.name == "tests":
|
|
27
|
+
continue
|
|
28
|
+
text = p.read_text()
|
|
29
|
+
for needle, msg in DANGER.items():
|
|
30
|
+
if needle in text:
|
|
31
|
+
complaints.append(f"{msg} in {p.name}")
|
|
32
|
+
if SECRET.search(text):
|
|
33
|
+
complaints.append(f"A_SECRET hard-coded credential in {p.name}")
|
|
34
|
+
# bare except = hiding failure, which grumpy hates
|
|
35
|
+
try:
|
|
36
|
+
for node in ast.walk(ast.parse(text)):
|
|
37
|
+
if isinstance(node, ast.ExceptHandler) and node.type is None:
|
|
38
|
+
complaints.append(f"A_BARE_EXCEPT swallowed error in {p.name}")
|
|
39
|
+
except SyntaxError:
|
|
40
|
+
complaints.append(f"A_SYNTAX {p.name} does not parse")
|
|
41
|
+
if require_tests:
|
|
42
|
+
test_files = [p for p in py if p.name.startswith("test_") or p.parent.name == "tests"]
|
|
43
|
+
if not test_files:
|
|
44
|
+
complaints.append("A_NO_PROOF no tests supplied — prove it works")
|
|
45
|
+
return (len(complaints) == 0, complaints)
|
forgeline/gates/judge.py
ADDED
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
"""Judge Agent — directory & interface consistency (executable, LLM-free).
|
|
2
|
+
Checks the filled code matches the SSAT scaffold: no stubs left, every
|
|
3
|
+
declared function implemented, imports resolve within the tree."""
|
|
4
|
+
from __future__ import annotations
|
|
5
|
+
import ast
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from ..ssat import check_erosion
|
|
8
|
+
|
|
9
|
+
def judge_consistency(ssat: dict, src_dir: Path) -> tuple[bool, list[str]]:
|
|
10
|
+
findings = []
|
|
11
|
+
src_dir = Path(src_dir)
|
|
12
|
+
for mod in ssat.get("modules", []):
|
|
13
|
+
p = src_dir/mod["path"]
|
|
14
|
+
if not p.exists():
|
|
15
|
+
findings.append(f"J_MISSING {mod['path']}"); continue
|
|
16
|
+
text = p.read_text()
|
|
17
|
+
if "raise NotImplementedError" in text or "# FILL" in text or "TODO" in text:
|
|
18
|
+
findings.append(f"J_STUB unfilled body in {mod['path']}")
|
|
19
|
+
try:
|
|
20
|
+
ast.parse(text)
|
|
21
|
+
except SyntaxError as e:
|
|
22
|
+
findings.append(f"J_SYNTAX {mod['path']}: {e}")
|
|
23
|
+
# structural erosion is a judge concern too
|
|
24
|
+
for v in check_erosion(ssat, src_dir):
|
|
25
|
+
findings.append(f"J_ARCH {v.code} {v.message}")
|
|
26
|
+
return (len(findings) == 0, findings)
|
|
@@ -0,0 +1,134 @@
|
|
|
1
|
+
"""Deep QA audit — stricter than the grumpy adversary's heuristics. Scores
|
|
2
|
+
generated code on coverage-intent, cyclomatic complexity, security surface,
|
|
3
|
+
and documentation, producing a QA grade that gates shipping. This is the
|
|
4
|
+
'stricter QA audit' layer: quantitative, thresholded, receipted."""
|
|
5
|
+
from __future__ import annotations
|
|
6
|
+
import ast, re
|
|
7
|
+
from dataclasses import dataclass, field
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
|
|
10
|
+
@dataclass
|
|
11
|
+
class QAReport:
|
|
12
|
+
coverage_intent: float = 0.0 # ratio of public funcs with a matching test
|
|
13
|
+
max_complexity: int = 0 # highest cyclomatic complexity found
|
|
14
|
+
security_score: int = 100 # 100 = clean; deductions per finding
|
|
15
|
+
doc_ratio: float = 0.0 # public funcs with docstrings
|
|
16
|
+
grade: str = "F"
|
|
17
|
+
findings: list = field(default_factory=list)
|
|
18
|
+
metrics: dict = field(default_factory=dict)
|
|
19
|
+
function_metrics: list = field(default_factory=list)
|
|
20
|
+
|
|
21
|
+
@property
|
|
22
|
+
def passed(self) -> bool:
|
|
23
|
+
return self.grade in ("A", "B") and self.security_score >= 80
|
|
24
|
+
|
|
25
|
+
@property
|
|
26
|
+
def attribution(self):
|
|
27
|
+
from ..attribution import Attribution, FailureClass, UnitResult
|
|
28
|
+
units = []
|
|
29
|
+
for metric in self.function_metrics:
|
|
30
|
+
failures = []
|
|
31
|
+
failure_class = None
|
|
32
|
+
if metric["complexity"] > 10:
|
|
33
|
+
failures.append(f"complexity={metric['complexity']}; threshold=10")
|
|
34
|
+
failure_class = FailureClass.COMPLEXITY_EXCEEDED
|
|
35
|
+
if not metric["tested"]:
|
|
36
|
+
failures.append("coverage_intent=0; required=1")
|
|
37
|
+
failure_class = failure_class or FailureClass.INCONSISTENT_LOGIC
|
|
38
|
+
units.append(UnitResult(
|
|
39
|
+
unit=f"qa_audit:{metric['function']}",
|
|
40
|
+
stage="qa_audit",
|
|
41
|
+
passed=not failures,
|
|
42
|
+
evidence="metrics within thresholds" if not failures else "; ".join(failures),
|
|
43
|
+
failure_class=failure_class,
|
|
44
|
+
))
|
|
45
|
+
if not units:
|
|
46
|
+
units.append(UnitResult(
|
|
47
|
+
"qa_audit:<no-public-functions>", "qa_audit", False,
|
|
48
|
+
"no public functions were available to grade",
|
|
49
|
+
FailureClass.STUB_UNFILLED,
|
|
50
|
+
))
|
|
51
|
+
return Attribution("qa_audit", len(units), sum(unit.passed for unit in units), units)
|
|
52
|
+
|
|
53
|
+
def _complexity(node: ast.FunctionDef) -> int:
|
|
54
|
+
"""Cyclomatic complexity: 1 + branch points."""
|
|
55
|
+
c = 1
|
|
56
|
+
for n in ast.walk(node):
|
|
57
|
+
if isinstance(n, (ast.If, ast.For, ast.While, ast.ExceptHandler, ast.With, ast.Assert)):
|
|
58
|
+
c += 1
|
|
59
|
+
elif isinstance(n, ast.BoolOp):
|
|
60
|
+
c += len(n.values) - 1
|
|
61
|
+
elif isinstance(n, ast.IfExp):
|
|
62
|
+
c += 1
|
|
63
|
+
return c
|
|
64
|
+
|
|
65
|
+
SEC_PATTERNS = {
|
|
66
|
+
r"\beval\(": ("CRITICAL", 40, "eval() — arbitrary code execution"),
|
|
67
|
+
r"\bexec\(": ("CRITICAL", 40, "exec() — arbitrary code execution"),
|
|
68
|
+
r"shell\s*=\s*True": ("HIGH", 25, "shell=True — command injection surface"),
|
|
69
|
+
r"pickle\.loads?\(": ("HIGH", 20, "pickle — unsafe deserialization"),
|
|
70
|
+
r"verify\s*=\s*False": ("HIGH", 20, "TLS verification disabled"),
|
|
71
|
+
r"(?i)(password|secret|api_key|token)\s*=\s*['\"][^'\"]{8,}": ("CRITICAL", 40, "hard-coded credential"),
|
|
72
|
+
r"md5\(": ("MEDIUM", 10, "MD5 — weak hash"),
|
|
73
|
+
r"random\.random\(\)": ("LOW", 5, "non-cryptographic randomness"),
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
def qa_audit(src_dir: Path) -> QAReport:
|
|
77
|
+
src_dir = Path(src_dir)
|
|
78
|
+
r = QAReport()
|
|
79
|
+
code_files = [p for p in src_dir.rglob("*.py")
|
|
80
|
+
if not p.name.startswith("test_") and p.parent.name != "tests"]
|
|
81
|
+
test_files = [p for p in src_dir.rglob("*.py")
|
|
82
|
+
if p.name.startswith("test_") or p.parent.name == "tests"]
|
|
83
|
+
all_test_text = "\n".join(p.read_text() for p in test_files)
|
|
84
|
+
|
|
85
|
+
public_funcs, tested, documented, complexities = [], 0, 0, []
|
|
86
|
+
for p in code_files:
|
|
87
|
+
try:
|
|
88
|
+
tree = ast.parse(p.read_text())
|
|
89
|
+
except SyntaxError as e:
|
|
90
|
+
r.findings.append(f"QA_SYNTAX {p.name}: {e}"); continue
|
|
91
|
+
for node in ast.walk(tree):
|
|
92
|
+
if isinstance(node, ast.FunctionDef) and not node.name.startswith("_"):
|
|
93
|
+
public_funcs.append(node.name)
|
|
94
|
+
# coverage-intent: is the function name referenced in tests?
|
|
95
|
+
if re.search(rf"\b{re.escape(node.name)}\b", all_test_text):
|
|
96
|
+
tested += 1
|
|
97
|
+
if ast.get_docstring(node):
|
|
98
|
+
documented += 1
|
|
99
|
+
cx = _complexity(node)
|
|
100
|
+
complexities.append((node.name, cx))
|
|
101
|
+
r.function_metrics.append({
|
|
102
|
+
"function": f"{p.name}:{node.name}",
|
|
103
|
+
"complexity": cx,
|
|
104
|
+
"tested": bool(re.search(rf"\b{re.escape(node.name)}\b", all_test_text)),
|
|
105
|
+
"documented": bool(ast.get_docstring(node)),
|
|
106
|
+
})
|
|
107
|
+
# security scan
|
|
108
|
+
text = p.read_text()
|
|
109
|
+
for pat, (sev, deduct, msg) in SEC_PATTERNS.items():
|
|
110
|
+
if re.search(pat, text):
|
|
111
|
+
r.security_score -= deduct
|
|
112
|
+
r.findings.append(f"QA_SEC[{sev}] {msg} in {p.name}")
|
|
113
|
+
|
|
114
|
+
n = len(public_funcs) or 1
|
|
115
|
+
r.coverage_intent = round(tested / n, 2)
|
|
116
|
+
r.doc_ratio = round(documented / n, 2)
|
|
117
|
+
r.max_complexity = max((c for _, c in complexities), default=0)
|
|
118
|
+
r.security_score = max(r.security_score, 0)
|
|
119
|
+
for name, cx in complexities:
|
|
120
|
+
if cx > 10:
|
|
121
|
+
r.findings.append(f"QA_COMPLEXITY {name}() complexity {cx} > 10 — refactor")
|
|
122
|
+
|
|
123
|
+
# composite grade
|
|
124
|
+
score = 0
|
|
125
|
+
score += 35 * r.coverage_intent
|
|
126
|
+
score += 25 * (1 if r.max_complexity <= 10 else max(0, 1 - (r.max_complexity-10)/10))
|
|
127
|
+
score += 25 * (r.security_score / 100)
|
|
128
|
+
score += 15 * r.doc_ratio
|
|
129
|
+
r.metrics = {"coverage_intent": r.coverage_intent, "max_complexity": r.max_complexity,
|
|
130
|
+
"security_score": r.security_score, "doc_ratio": r.doc_ratio,
|
|
131
|
+
"composite": round(score, 1)}
|
|
132
|
+
r.grade = ("A" if score >= 85 else "B" if score >= 70 else
|
|
133
|
+
"C" if score >= 55 else "D" if score >= 40 else "F")
|
|
134
|
+
return r
|
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
"""Reverse-classical test verification.
|
|
2
|
+
|
|
3
|
+
Before ForgeLine trusts a smoke check against the real implementation, it runs
|
|
4
|
+
the same behavioral check against the generated SSAT scaffold. A behavioral
|
|
5
|
+
test must fail on that empty stub. If it passes, the test is hollow: it asserts
|
|
6
|
+
nothing the implementation provides.
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import shutil
|
|
11
|
+
import tempfile
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
|
|
14
|
+
from forgeline.attribution import Attribution, FailureClass, GateResult, UnitResult
|
|
15
|
+
from forgeline.gates.runtime_smoke import _load_manifest, _run_check
|
|
16
|
+
from forgeline.ssat import load_ssat, scaffold_from_ssat
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def materialize_stub_root(ssat_path: Path) -> Path:
|
|
20
|
+
"""Regenerate the SSAT scaffold into an isolated temp root.
|
|
21
|
+
|
|
22
|
+
The production tree is never touched. Reusing `scaffold_from_ssat` is the
|
|
23
|
+
guarantee: the mutant is byte-identical to what the normal SCAFFOLDED state
|
|
24
|
+
would produce for the same SSAT.
|
|
25
|
+
"""
|
|
26
|
+
tmp = Path(tempfile.mkdtemp(prefix="forge-stub-"))
|
|
27
|
+
scaffold_from_ssat(load_ssat(ssat_path), tmp)
|
|
28
|
+
return tmp
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def verify_tests(root: Path, feature: str, ssat_path: Path) -> GateResult:
|
|
32
|
+
root = Path(root)
|
|
33
|
+
try:
|
|
34
|
+
checks = _load_manifest(root, feature)
|
|
35
|
+
except ValueError as exc:
|
|
36
|
+
attr = Attribution("verify_tests", 1, 0, [
|
|
37
|
+
UnitResult(
|
|
38
|
+
unit="verify_tests:manifest",
|
|
39
|
+
stage="verify_tests",
|
|
40
|
+
passed=False,
|
|
41
|
+
evidence=str(exc),
|
|
42
|
+
failure_class=FailureClass.HOLLOW_MANIFEST,
|
|
43
|
+
)
|
|
44
|
+
])
|
|
45
|
+
return GateResult(False, attr)
|
|
46
|
+
|
|
47
|
+
if checks is None:
|
|
48
|
+
attr = Attribution("verify_tests", 1, 0, [
|
|
49
|
+
UnitResult(
|
|
50
|
+
unit="verify_tests:manifest",
|
|
51
|
+
stage="verify_tests",
|
|
52
|
+
passed=False,
|
|
53
|
+
evidence=f"no smoke manifest at smoke/{feature}.json; nothing to verify",
|
|
54
|
+
failure_class=FailureClass.HOLLOW_MANIFEST,
|
|
55
|
+
)
|
|
56
|
+
])
|
|
57
|
+
return GateResult(False, attr)
|
|
58
|
+
|
|
59
|
+
if not checks:
|
|
60
|
+
attr = Attribution("verify_tests", 1, 0, [
|
|
61
|
+
UnitResult(
|
|
62
|
+
unit="verify_tests:manifest",
|
|
63
|
+
stage="verify_tests",
|
|
64
|
+
passed=False,
|
|
65
|
+
evidence="smoke manifest declares no checks",
|
|
66
|
+
failure_class=FailureClass.HOLLOW_MANIFEST,
|
|
67
|
+
)
|
|
68
|
+
])
|
|
69
|
+
return GateResult(False, attr)
|
|
70
|
+
|
|
71
|
+
if all(not check.must_fail_on_stub for check in checks):
|
|
72
|
+
units = [
|
|
73
|
+
UnitResult(
|
|
74
|
+
unit=f"verify_tests:{check.name}",
|
|
75
|
+
stage="verify_tests",
|
|
76
|
+
passed=False,
|
|
77
|
+
evidence="every check is exempt; manifest verifies no behavior",
|
|
78
|
+
failure_class=FailureClass.HOLLOW_MANIFEST,
|
|
79
|
+
)
|
|
80
|
+
for check in checks
|
|
81
|
+
]
|
|
82
|
+
return GateResult(False, Attribution("verify_tests", len(units), 0, units))
|
|
83
|
+
|
|
84
|
+
stub_root = materialize_stub_root(ssat_path)
|
|
85
|
+
try:
|
|
86
|
+
units: list[UnitResult] = []
|
|
87
|
+
for check in checks:
|
|
88
|
+
unit = f"verify_tests:{check.name}"
|
|
89
|
+
if not check.must_fail_on_stub:
|
|
90
|
+
units.append(UnitResult(
|
|
91
|
+
unit=unit,
|
|
92
|
+
stage="verify_tests",
|
|
93
|
+
passed=True,
|
|
94
|
+
evidence="exempt: declared structural check",
|
|
95
|
+
failure_class=None,
|
|
96
|
+
))
|
|
97
|
+
continue
|
|
98
|
+
|
|
99
|
+
result = _run_check(check, stub_root)
|
|
100
|
+
hollow = result.passed
|
|
101
|
+
units.append(UnitResult(
|
|
102
|
+
unit=unit,
|
|
103
|
+
stage="verify_tests",
|
|
104
|
+
passed=not hollow,
|
|
105
|
+
evidence=(
|
|
106
|
+
"check PASSED against an empty stub; it asserts nothing the "
|
|
107
|
+
"implementation provides"
|
|
108
|
+
if hollow
|
|
109
|
+
else f"correctly failed on stub: {result.reason}"
|
|
110
|
+
),
|
|
111
|
+
failure_class=FailureClass.HOLLOW_TEST if hollow else None,
|
|
112
|
+
))
|
|
113
|
+
finally:
|
|
114
|
+
shutil.rmtree(stub_root, ignore_errors=True)
|
|
115
|
+
|
|
116
|
+
attr = Attribution("verify_tests", len(units), sum(unit.passed for unit in units), units)
|
|
117
|
+
return GateResult(attr.rate == 1.0, attr)
|
|
118
|
+
|
|
@@ -0,0 +1,194 @@
|
|
|
1
|
+
"""forgeline runtime smoke gate — behavior-by-inspection before ship.
|
|
2
|
+
|
|
3
|
+
The rest of ForgeLine verifies code against *specifications*: the judge checks
|
|
4
|
+
consistency, the QA audit grades static quality, the intent thread proves the
|
|
5
|
+
code honors the sealed envelope. All of that is correctness-*by-construction*.
|
|
6
|
+
|
|
7
|
+
None of it answers the question a per-PR preview deployment answers: **does the
|
|
8
|
+
built thing actually RUN and behave correctly when executed?** A change can pass
|
|
9
|
+
every static gate and still crash on import, throw at runtime, or produce the
|
|
10
|
+
wrong output. This gate closes that gap at the right scale for a solo builder:
|
|
11
|
+
it runs the artifact against declared behavioral checks and blocks ship on any
|
|
12
|
+
runtime failure — without the cost of ephemeral per-PR environments.
|
|
13
|
+
|
|
14
|
+
Design contract:
|
|
15
|
+
- Deterministic pass/fail: a check either ran green or it didn't.
|
|
16
|
+
- Isolated: checks run in a subprocess with a timeout, so a hang or crash in
|
|
17
|
+
the built code cannot take down the orchestrator.
|
|
18
|
+
- Declarative: behavioral checks live in a `smoke/` manifest beside the spec,
|
|
19
|
+
so what "correct runtime behavior" means is itself a reviewed artifact.
|
|
20
|
+
- Fail-closed: no manifest, or a manifest that references nothing runnable,
|
|
21
|
+
is a BLOCK — you cannot ship unverified runtime behavior by omission.
|
|
22
|
+
"""
|
|
23
|
+
from __future__ import annotations
|
|
24
|
+
import json
|
|
25
|
+
import subprocess
|
|
26
|
+
import sys
|
|
27
|
+
import time
|
|
28
|
+
from dataclasses import dataclass, field
|
|
29
|
+
from pathlib import Path
|
|
30
|
+
from forgeline.attribution import Attribution, FailureClass, GateResult, UnitResult
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
@dataclass
|
|
34
|
+
class SmokeCheck:
|
|
35
|
+
name: str
|
|
36
|
+
kind: str # "command" | "python"
|
|
37
|
+
run: str # shell command, or python snippet/file
|
|
38
|
+
expect_exit: int = 0
|
|
39
|
+
expect_stdout: str | None = None # substring that must appear
|
|
40
|
+
timeout_s: int = 30
|
|
41
|
+
must_fail_on_stub: bool = True
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
@dataclass
|
|
45
|
+
class SmokeResult:
|
|
46
|
+
name: str
|
|
47
|
+
passed: bool
|
|
48
|
+
reason: str
|
|
49
|
+
duration_ms: int = 0
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
@dataclass
|
|
53
|
+
class SmokeReport:
|
|
54
|
+
results: list[SmokeResult] = field(default_factory=list)
|
|
55
|
+
manifest_found: bool = True
|
|
56
|
+
|
|
57
|
+
@property
|
|
58
|
+
def ok(self) -> bool:
|
|
59
|
+
return self.manifest_found and bool(self.results) and all(r.passed for r in self.results)
|
|
60
|
+
|
|
61
|
+
@property
|
|
62
|
+
def failures(self):
|
|
63
|
+
return [r for r in self.results if not r.passed]
|
|
64
|
+
|
|
65
|
+
def add(self, name, passed, reason, duration_ms=0):
|
|
66
|
+
self.results.append(SmokeResult(name, passed, reason, duration_ms))
|
|
67
|
+
|
|
68
|
+
@property
|
|
69
|
+
def attribution(self) -> Attribution:
|
|
70
|
+
units = []
|
|
71
|
+
for result in self.results:
|
|
72
|
+
failure_class = None
|
|
73
|
+
reason = result.reason.lower()
|
|
74
|
+
if not result.passed:
|
|
75
|
+
if "timed out" in reason:
|
|
76
|
+
failure_class = FailureClass.RUNTIME_TIMEOUT
|
|
77
|
+
elif "expected stdout" in reason:
|
|
78
|
+
failure_class = FailureClass.WRONG_OUTPUT
|
|
79
|
+
else:
|
|
80
|
+
failure_class = FailureClass.RUNTIME_CRASH
|
|
81
|
+
units.append(UnitResult(
|
|
82
|
+
unit=f"smoke:{result.name}",
|
|
83
|
+
stage="smoke",
|
|
84
|
+
passed=result.passed,
|
|
85
|
+
evidence=result.reason,
|
|
86
|
+
failure_class=failure_class,
|
|
87
|
+
))
|
|
88
|
+
return Attribution("smoke", len(units), sum(unit.passed for unit in units), units)
|
|
89
|
+
|
|
90
|
+
@property
|
|
91
|
+
def gate_result(self) -> GateResult:
|
|
92
|
+
attr = self.attribution
|
|
93
|
+
return GateResult(attr.n_checked > 0 and attr.rate == 1.0, attr)
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def _load_manifest(root: Path, feature: str) -> list[SmokeCheck] | None:
|
|
97
|
+
"""A smoke manifest is JSON at smoke/<feature>.json (or smoke/smoke.json).
|
|
98
|
+
Each entry declares one behavioral check. Missing manifest -> None (fail-closed
|
|
99
|
+
upstream)."""
|
|
100
|
+
candidates = [
|
|
101
|
+
Path(root) / "smoke" / f"{feature}.json",
|
|
102
|
+
Path(root) / "smoke" / "smoke.json",
|
|
103
|
+
]
|
|
104
|
+
for p in candidates:
|
|
105
|
+
if p.exists():
|
|
106
|
+
try:
|
|
107
|
+
data = json.loads(p.read_text())
|
|
108
|
+
except json.JSONDecodeError as e:
|
|
109
|
+
raise ValueError(f"smoke manifest {p} is not valid JSON: {e}")
|
|
110
|
+
checks = []
|
|
111
|
+
for entry in data.get("checks", []):
|
|
112
|
+
checks.append(SmokeCheck(
|
|
113
|
+
name=entry["name"],
|
|
114
|
+
kind=entry.get("kind", "command"),
|
|
115
|
+
run=entry["run"],
|
|
116
|
+
expect_exit=entry.get("expect_exit", 0),
|
|
117
|
+
expect_stdout=entry.get("expect_stdout"),
|
|
118
|
+
timeout_s=entry.get("timeout_s", 30),
|
|
119
|
+
must_fail_on_stub=entry.get("must_fail_on_stub", True),
|
|
120
|
+
))
|
|
121
|
+
return checks
|
|
122
|
+
return None
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def _run_check(check: SmokeCheck, cwd: Path) -> SmokeResult:
|
|
126
|
+
"""Execute one check in an isolated subprocess with a hard timeout."""
|
|
127
|
+
t0 = time.monotonic()
|
|
128
|
+
if check.kind == "python":
|
|
129
|
+
cmd = [sys.executable, "-c", check.run]
|
|
130
|
+
else:
|
|
131
|
+
cmd = check.run # shell command string
|
|
132
|
+
try:
|
|
133
|
+
proc = subprocess.run(
|
|
134
|
+
cmd,
|
|
135
|
+
cwd=str(cwd),
|
|
136
|
+
shell=(check.kind != "python"),
|
|
137
|
+
capture_output=True,
|
|
138
|
+
text=True,
|
|
139
|
+
timeout=check.timeout_s,
|
|
140
|
+
)
|
|
141
|
+
except subprocess.TimeoutExpired:
|
|
142
|
+
ms = int((time.monotonic() - t0) * 1000)
|
|
143
|
+
return SmokeResult(check.name, False,
|
|
144
|
+
f"timed out after {check.timeout_s}s — runtime hang", ms)
|
|
145
|
+
except (OSError, ValueError) as e:
|
|
146
|
+
ms = int((time.monotonic() - t0) * 1000)
|
|
147
|
+
return SmokeResult(check.name, False, f"could not execute: {e}", ms)
|
|
148
|
+
|
|
149
|
+
ms = int((time.monotonic() - t0) * 1000)
|
|
150
|
+
# exit-code check
|
|
151
|
+
if proc.returncode != check.expect_exit:
|
|
152
|
+
tail = (proc.stderr or proc.stdout or "").strip().splitlines()[-3:]
|
|
153
|
+
return SmokeResult(check.name, False,
|
|
154
|
+
f"exit {proc.returncode} != expected {check.expect_exit}; "
|
|
155
|
+
f"last: {' | '.join(tail)[:200]}", ms)
|
|
156
|
+
# stdout-substring check
|
|
157
|
+
if check.expect_stdout is not None and check.expect_stdout not in (proc.stdout or ""):
|
|
158
|
+
return SmokeResult(check.name, False,
|
|
159
|
+
f"expected stdout to contain {check.expect_stdout!r}, not found", ms)
|
|
160
|
+
return SmokeResult(check.name, True, "ran green", ms)
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def runtime_smoke(root: Path, feature: str) -> SmokeReport:
|
|
164
|
+
"""Run all declared behavioral checks for a feature. Fail-closed: no manifest,
|
|
165
|
+
empty manifest, or any failing check blocks ship."""
|
|
166
|
+
rep = SmokeReport()
|
|
167
|
+
root = Path(root)
|
|
168
|
+
try:
|
|
169
|
+
checks = _load_manifest(root, feature)
|
|
170
|
+
except ValueError as e:
|
|
171
|
+
rep.manifest_found = True
|
|
172
|
+
rep.add("manifest", False, str(e))
|
|
173
|
+
return rep
|
|
174
|
+
if checks is None:
|
|
175
|
+
rep.manifest_found = False
|
|
176
|
+
rep.add("manifest", False,
|
|
177
|
+
f"no smoke manifest at smoke/{feature}.json — cannot verify runtime "
|
|
178
|
+
f"behavior. Declare at least one behavioral check before ship.")
|
|
179
|
+
return rep
|
|
180
|
+
if not checks:
|
|
181
|
+
rep.add("manifest", False, "smoke manifest present but declares no checks — "
|
|
182
|
+
"runtime behavior would ship unverified.")
|
|
183
|
+
return rep
|
|
184
|
+
for c in checks:
|
|
185
|
+
rep.results.append(_run_check(c, root))
|
|
186
|
+
return rep
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def smoke_report_lines(rep: SmokeReport) -> list[str]:
|
|
190
|
+
out = []
|
|
191
|
+
for r in rep.results:
|
|
192
|
+
mark = "PASS" if r.passed else "FAIL"
|
|
193
|
+
out.append(f"[{mark}] {r.name} ({r.duration_ms}ms): {r.reason}")
|
|
194
|
+
return out
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
"""Skill-check gate — ensures the AGENT EXPERIENCE stays high: the run
|
|
2
|
+
produced the receipts, the state machine advanced legally, and the skill
|
|
3
|
+
memory recorded a lesson. Keeps the factory itself from eroding."""
|
|
4
|
+
from __future__ import annotations
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
|
|
7
|
+
def skill_check(root: Path, feature: str) -> tuple[bool, list[str]]:
|
|
8
|
+
findings = []
|
|
9
|
+
fd = Path(root)/".forge"/feature
|
|
10
|
+
if not (fd/"receipts.jsonl").exists() or not (fd/"receipts.jsonl").read_text().strip():
|
|
11
|
+
findings.append("S_NO_RECEIPTS run produced no receipts")
|
|
12
|
+
if not (fd/"state.json").exists():
|
|
13
|
+
findings.append("S_NO_STATE run has no state record")
|
|
14
|
+
return (len(findings) == 0, findings)
|