code-factory-2-forge 0.6.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
forgeline/demo.py ADDED
@@ -0,0 +1,83 @@
1
+ """forge demo - a headless walk through refine loop + skill flywheel."""
2
+ from __future__ import annotations
3
+
4
+ import shutil
5
+ import tempfile
6
+ import time
7
+ from pathlib import Path
8
+
9
+ C = {"g": "\033[92m", "c": "\033[96m", "y": "\033[93m", "d": "\033[2m", "b": "\033[1m", "x": "\033[0m"}
10
+
11
+
12
+ def p(message: str, sleep: float = 0.02) -> None:
13
+ print(message.encode("ascii", "replace").decode("ascii"))
14
+ time.sleep(sleep)
15
+
16
+
17
+ def run() -> None:
18
+ from forgeline.orchestrator import Orchestrator
19
+ from forgeline.skill_memory import lessons_for
20
+ from forgeline.states import State
21
+
22
+ root = Path(tempfile.mkdtemp())
23
+ (root / "skills").mkdir()
24
+ ssat_src = Path(__file__).resolve().parents[1] / "examples" / "notifier.ssat.yaml"
25
+ ssat = root / "notifier.ssat.yaml"
26
+ shutil.copy(ssat_src, ssat)
27
+
28
+ p(f"\n{C['b']}* ForgeLine - autonomous factory outer loop{C['x']}")
29
+ p(f"{C['d']} intent -> SSAT -> scaffold -> fill -> adversarial review -> ship{C['x']}\n")
30
+
31
+ orchestrator = Orchestrator(root, "notifier")
32
+ orchestrator.store.set_state(State.ARCHITECTED)
33
+ p(f"{C['c']}[1] architect{C['x']} - scaffolding signatures from architecture-as-code")
34
+ orchestrator.architect(ssat)
35
+ p(f" {C['g']}OK{C['x']} 2 modules scaffolded with valid imports\n")
36
+
37
+ slices = root / "slices" / "notifier"
38
+ (slices / "formatter.py").write_text(
39
+ 'def format_message(event: dict) -> str:\n return str(eval(event.get("expr","1")))\n',
40
+ encoding="utf-8",
41
+ )
42
+ (slices / "sender.py").write_text(
43
+ "from slices.notifier.formatter import format_message\n"
44
+ "def send(event: dict, channel: str) -> bool:\n"
45
+ " return bool(format_message(event) and channel)\n",
46
+ encoding="utf-8",
47
+ )
48
+ orchestrator.store.set_state(State.FILLED)
49
+ p(f"{C['c']}[2] fill (attempt 1){C['x']} - agent used eval(), shipped no tests")
50
+ review = orchestrator.review(ssat)
51
+ p(f" {C['y']}BLOCKED: grumpy adversary findings:{C['x']}")
52
+ for finding in review["findings"]:
53
+ p(f" - {finding}")
54
+ p(f" {C['d']}lesson recorded to skill memory{C['x']}\n")
55
+
56
+ p(f"{C['c']}[3] lessons injected into next context:{C['x']}")
57
+ for lesson in lessons_for(root, "fill"):
58
+ p(f" [{lesson['failure_code']} x{lesson['count']}] {lesson['fix'][:60]}")
59
+
60
+ (slices / "formatter.py").write_text(
61
+ "def format_message(event: dict) -> str:\n"
62
+ " return f\"{event.get('kind','x')}: {event.get('text','')}\"\n",
63
+ encoding="utf-8",
64
+ )
65
+ (root / "tests").mkdir(exist_ok=True)
66
+ (root / "tests" / "test_notifier.py").write_text(
67
+ "def test_ok():\n assert True\n",
68
+ encoding="utf-8",
69
+ )
70
+ p(f"\n{C['c']}[4] fill (attempt 2){C['x']} - eval removed, tests supplied")
71
+ review2 = orchestrator.review(ssat)
72
+ if review2["reviewed"]:
73
+ p(f" {C['g']}OK judge + grumpy + arch erosion all pass{C['x']}")
74
+ orchestrator.arch_gate(ssat)
75
+ orchestrator.ship()
76
+ p(f" {C['g']}OK architecture CI gate pass -> SHIPPED{C['x']}\n")
77
+
78
+ p(f"{C['d']}The factory caught its own bad output, learned, and shipped clean.")
79
+ p(f" Wire it in: forge agent claude | forge agent codex{C['x']}\n")
80
+
81
+
82
+ if __name__ == "__main__":
83
+ run()
@@ -0,0 +1,44 @@
1
+ """forge demo-learning — shows the recursive loop: a failure recurs, gets
2
+ promoted into active policy, then that policy catches it on the next run and
3
+ is marked validated. The factory improving its own QA."""
4
+ from __future__ import annotations
5
+ import tempfile, time
6
+ from pathlib import Path
7
+ from .learning import LearningKernel
8
+ from .skill_memory import record_lesson, lessons_for
9
+
10
+ C = {"g":"\033[92m","c":"\033[96m","y":"\033[93m","d":"\033[2m","b":"\033[1m","x":"\033[0m"}
11
+ def p(m): print(m); time.sleep(0.02)
12
+
13
+ def run():
14
+ root = Path(tempfile.mkdtemp()); (root/"skills").mkdir()
15
+ k = LearningKernel(root, promote_threshold=3)
16
+ p(f"\n{C['b']}✦ ForgeLine — recursive learning loop{C['x']}")
17
+ p(f"{C['d']} observe → promote → validate → self-prune{C['x']}\n")
18
+
19
+ p(f"{C['c']}[1] observe{C['x']} — the same flaw (eval) appears in 3 runs")
20
+ for i in range(3):
21
+ record_lesson(root, phase="fill", failure_code="A_EVAL", fix="never use eval()", feature=f"run{i}")
22
+ p(f" {C['y']}▲{C['x']} run {i+1}: A_EVAL recorded")
23
+
24
+ p(f"\n{C['c']}[2] promote{C['x']} — seen ≥3× → becomes an ENFORCED constraint")
25
+ promoted = k.promote(lessons_for(root, "fill"))
26
+ p(f" {C['g']}✓{C['x']} promoted to active policy: {promoted}")
27
+
28
+ p(f"\n{C['c']}[3] validate{C['x']} — next run, the policy CATCHES the same flaw")
29
+ prevented = k.enforce(["A_EVAL"])
30
+ p(f" {C['g']}✓{C['x']} A_EVAL caught by learned policy → validated")
31
+
32
+ p(f"\n{C['c']}[4] self-prune{C['x']} — a stale rule that never fires goes to probation")
33
+ for i in range(3):
34
+ record_lesson(root, phase="fill", failure_code="A_OLD", fix="obsolete rule", feature="x")
35
+ k.promote(lessons_for(root, "fill"))
36
+ eff = k.audit_effectiveness()
37
+ p(f" {C['d']}validated: {eff['validated']} · probation: {eff['probation']}{C['x']}")
38
+
39
+ s = k.policy_summary()
40
+ p(f"\n{C['d']}Active policy v{s['version']} — the factory now enforces what it learned.")
41
+ p(f" Every future build is checked against its own accumulated experience.{C['x']}\n")
42
+
43
+ if __name__ == "__main__":
44
+ run()
@@ -0,0 +1,4 @@
1
+ from .judge import judge_consistency
2
+ from .adversary import grumpy_review
3
+ from .skill_check import skill_check
4
+ __all__ = ["judge_consistency", "grumpy_review", "skill_check"]
@@ -0,0 +1,45 @@
1
+ """The grumpy adversary — assumes the code is broken/insecure and makes the
2
+ generator prove otherwise. Executable heuristics (no LLM needed to be
3
+ useful); an optional LLM adversary can be layered behind the same interface.
4
+ It does NOT need to be right — it needs to force proof (tests + arch completeness)."""
5
+ from __future__ import annotations
6
+ import ast, re
7
+ from pathlib import Path
8
+
9
+ DANGER = {
10
+ "eval(": "A_EVAL arbitrary eval",
11
+ "exec(": "A_EXEC arbitrary exec",
12
+ "os.system": "A_SHELL shell execution",
13
+ "subprocess": "A_SUBPROC subprocess use",
14
+ "pickle.load": "A_PICKLE unsafe deserialization",
15
+ "verify=False": "A_TLS TLS verification disabled",
16
+ "shell=True": "A_SHELL_TRUE shell=True injection surface",
17
+ }
18
+ SECRET = re.compile(r"(?i)(api[_-]?key|secret|password|token)\s*=\s*['\"][A-Za-z0-9/\+_-]{12,}['\"]")
19
+
20
+ def grumpy_review(src_dir: Path, require_tests: bool = True) -> tuple[bool, list[str]]:
21
+ """Returns (satisfied, complaints). Grumpy is satisfied only when it can
22
+ find NO obvious sin AND the generator supplied tests to prove correctness."""
23
+ src_dir = Path(src_dir); complaints = []
24
+ py = list(src_dir.rglob("*.py"))
25
+ for p in py:
26
+ if p.name.startswith("test_") or p.parent.name == "tests":
27
+ continue
28
+ text = p.read_text()
29
+ for needle, msg in DANGER.items():
30
+ if needle in text:
31
+ complaints.append(f"{msg} in {p.name}")
32
+ if SECRET.search(text):
33
+ complaints.append(f"A_SECRET hard-coded credential in {p.name}")
34
+ # bare except = hiding failure, which grumpy hates
35
+ try:
36
+ for node in ast.walk(ast.parse(text)):
37
+ if isinstance(node, ast.ExceptHandler) and node.type is None:
38
+ complaints.append(f"A_BARE_EXCEPT swallowed error in {p.name}")
39
+ except SyntaxError:
40
+ complaints.append(f"A_SYNTAX {p.name} does not parse")
41
+ if require_tests:
42
+ test_files = [p for p in py if p.name.startswith("test_") or p.parent.name == "tests"]
43
+ if not test_files:
44
+ complaints.append("A_NO_PROOF no tests supplied — prove it works")
45
+ return (len(complaints) == 0, complaints)
@@ -0,0 +1,26 @@
1
+ """Judge Agent — directory & interface consistency (executable, LLM-free).
2
+ Checks the filled code matches the SSAT scaffold: no stubs left, every
3
+ declared function implemented, imports resolve within the tree."""
4
+ from __future__ import annotations
5
+ import ast
6
+ from pathlib import Path
7
+ from ..ssat import check_erosion
8
+
9
+ def judge_consistency(ssat: dict, src_dir: Path) -> tuple[bool, list[str]]:
10
+ findings = []
11
+ src_dir = Path(src_dir)
12
+ for mod in ssat.get("modules", []):
13
+ p = src_dir/mod["path"]
14
+ if not p.exists():
15
+ findings.append(f"J_MISSING {mod['path']}"); continue
16
+ text = p.read_text()
17
+ if "raise NotImplementedError" in text or "# FILL" in text or "TODO" in text:
18
+ findings.append(f"J_STUB unfilled body in {mod['path']}")
19
+ try:
20
+ ast.parse(text)
21
+ except SyntaxError as e:
22
+ findings.append(f"J_SYNTAX {mod['path']}: {e}")
23
+ # structural erosion is a judge concern too
24
+ for v in check_erosion(ssat, src_dir):
25
+ findings.append(f"J_ARCH {v.code} {v.message}")
26
+ return (len(findings) == 0, findings)
@@ -0,0 +1,134 @@
1
+ """Deep QA audit — stricter than the grumpy adversary's heuristics. Scores
2
+ generated code on coverage-intent, cyclomatic complexity, security surface,
3
+ and documentation, producing a QA grade that gates shipping. This is the
4
+ 'stricter QA audit' layer: quantitative, thresholded, receipted."""
5
+ from __future__ import annotations
6
+ import ast, re
7
+ from dataclasses import dataclass, field
8
+ from pathlib import Path
9
+
10
+ @dataclass
11
+ class QAReport:
12
+ coverage_intent: float = 0.0 # ratio of public funcs with a matching test
13
+ max_complexity: int = 0 # highest cyclomatic complexity found
14
+ security_score: int = 100 # 100 = clean; deductions per finding
15
+ doc_ratio: float = 0.0 # public funcs with docstrings
16
+ grade: str = "F"
17
+ findings: list = field(default_factory=list)
18
+ metrics: dict = field(default_factory=dict)
19
+ function_metrics: list = field(default_factory=list)
20
+
21
+ @property
22
+ def passed(self) -> bool:
23
+ return self.grade in ("A", "B") and self.security_score >= 80
24
+
25
+ @property
26
+ def attribution(self):
27
+ from ..attribution import Attribution, FailureClass, UnitResult
28
+ units = []
29
+ for metric in self.function_metrics:
30
+ failures = []
31
+ failure_class = None
32
+ if metric["complexity"] > 10:
33
+ failures.append(f"complexity={metric['complexity']}; threshold=10")
34
+ failure_class = FailureClass.COMPLEXITY_EXCEEDED
35
+ if not metric["tested"]:
36
+ failures.append("coverage_intent=0; required=1")
37
+ failure_class = failure_class or FailureClass.INCONSISTENT_LOGIC
38
+ units.append(UnitResult(
39
+ unit=f"qa_audit:{metric['function']}",
40
+ stage="qa_audit",
41
+ passed=not failures,
42
+ evidence="metrics within thresholds" if not failures else "; ".join(failures),
43
+ failure_class=failure_class,
44
+ ))
45
+ if not units:
46
+ units.append(UnitResult(
47
+ "qa_audit:<no-public-functions>", "qa_audit", False,
48
+ "no public functions were available to grade",
49
+ FailureClass.STUB_UNFILLED,
50
+ ))
51
+ return Attribution("qa_audit", len(units), sum(unit.passed for unit in units), units)
52
+
53
+ def _complexity(node: ast.FunctionDef) -> int:
54
+ """Cyclomatic complexity: 1 + branch points."""
55
+ c = 1
56
+ for n in ast.walk(node):
57
+ if isinstance(n, (ast.If, ast.For, ast.While, ast.ExceptHandler, ast.With, ast.Assert)):
58
+ c += 1
59
+ elif isinstance(n, ast.BoolOp):
60
+ c += len(n.values) - 1
61
+ elif isinstance(n, ast.IfExp):
62
+ c += 1
63
+ return c
64
+
65
+ SEC_PATTERNS = {
66
+ r"\beval\(": ("CRITICAL", 40, "eval() — arbitrary code execution"),
67
+ r"\bexec\(": ("CRITICAL", 40, "exec() — arbitrary code execution"),
68
+ r"shell\s*=\s*True": ("HIGH", 25, "shell=True — command injection surface"),
69
+ r"pickle\.loads?\(": ("HIGH", 20, "pickle — unsafe deserialization"),
70
+ r"verify\s*=\s*False": ("HIGH", 20, "TLS verification disabled"),
71
+ r"(?i)(password|secret|api_key|token)\s*=\s*['\"][^'\"]{8,}": ("CRITICAL", 40, "hard-coded credential"),
72
+ r"md5\(": ("MEDIUM", 10, "MD5 — weak hash"),
73
+ r"random\.random\(\)": ("LOW", 5, "non-cryptographic randomness"),
74
+ }
75
+
76
+ def qa_audit(src_dir: Path) -> QAReport:
77
+ src_dir = Path(src_dir)
78
+ r = QAReport()
79
+ code_files = [p for p in src_dir.rglob("*.py")
80
+ if not p.name.startswith("test_") and p.parent.name != "tests"]
81
+ test_files = [p for p in src_dir.rglob("*.py")
82
+ if p.name.startswith("test_") or p.parent.name == "tests"]
83
+ all_test_text = "\n".join(p.read_text() for p in test_files)
84
+
85
+ public_funcs, tested, documented, complexities = [], 0, 0, []
86
+ for p in code_files:
87
+ try:
88
+ tree = ast.parse(p.read_text())
89
+ except SyntaxError as e:
90
+ r.findings.append(f"QA_SYNTAX {p.name}: {e}"); continue
91
+ for node in ast.walk(tree):
92
+ if isinstance(node, ast.FunctionDef) and not node.name.startswith("_"):
93
+ public_funcs.append(node.name)
94
+ # coverage-intent: is the function name referenced in tests?
95
+ if re.search(rf"\b{re.escape(node.name)}\b", all_test_text):
96
+ tested += 1
97
+ if ast.get_docstring(node):
98
+ documented += 1
99
+ cx = _complexity(node)
100
+ complexities.append((node.name, cx))
101
+ r.function_metrics.append({
102
+ "function": f"{p.name}:{node.name}",
103
+ "complexity": cx,
104
+ "tested": bool(re.search(rf"\b{re.escape(node.name)}\b", all_test_text)),
105
+ "documented": bool(ast.get_docstring(node)),
106
+ })
107
+ # security scan
108
+ text = p.read_text()
109
+ for pat, (sev, deduct, msg) in SEC_PATTERNS.items():
110
+ if re.search(pat, text):
111
+ r.security_score -= deduct
112
+ r.findings.append(f"QA_SEC[{sev}] {msg} in {p.name}")
113
+
114
+ n = len(public_funcs) or 1
115
+ r.coverage_intent = round(tested / n, 2)
116
+ r.doc_ratio = round(documented / n, 2)
117
+ r.max_complexity = max((c for _, c in complexities), default=0)
118
+ r.security_score = max(r.security_score, 0)
119
+ for name, cx in complexities:
120
+ if cx > 10:
121
+ r.findings.append(f"QA_COMPLEXITY {name}() complexity {cx} > 10 — refactor")
122
+
123
+ # composite grade
124
+ score = 0
125
+ score += 35 * r.coverage_intent
126
+ score += 25 * (1 if r.max_complexity <= 10 else max(0, 1 - (r.max_complexity-10)/10))
127
+ score += 25 * (r.security_score / 100)
128
+ score += 15 * r.doc_ratio
129
+ r.metrics = {"coverage_intent": r.coverage_intent, "max_complexity": r.max_complexity,
130
+ "security_score": r.security_score, "doc_ratio": r.doc_ratio,
131
+ "composite": round(score, 1)}
132
+ r.grade = ("A" if score >= 85 else "B" if score >= 70 else
133
+ "C" if score >= 55 else "D" if score >= 40 else "F")
134
+ return r
@@ -0,0 +1,118 @@
1
+ """Reverse-classical test verification.
2
+
3
+ Before ForgeLine trusts a smoke check against the real implementation, it runs
4
+ the same behavioral check against the generated SSAT scaffold. A behavioral
5
+ test must fail on that empty stub. If it passes, the test is hollow: it asserts
6
+ nothing the implementation provides.
7
+ """
8
+ from __future__ import annotations
9
+
10
+ import shutil
11
+ import tempfile
12
+ from pathlib import Path
13
+
14
+ from forgeline.attribution import Attribution, FailureClass, GateResult, UnitResult
15
+ from forgeline.gates.runtime_smoke import _load_manifest, _run_check
16
+ from forgeline.ssat import load_ssat, scaffold_from_ssat
17
+
18
+
19
+ def materialize_stub_root(ssat_path: Path) -> Path:
20
+ """Regenerate the SSAT scaffold into an isolated temp root.
21
+
22
+ The production tree is never touched. Reusing `scaffold_from_ssat` is the
23
+ guarantee: the mutant is byte-identical to what the normal SCAFFOLDED state
24
+ would produce for the same SSAT.
25
+ """
26
+ tmp = Path(tempfile.mkdtemp(prefix="forge-stub-"))
27
+ scaffold_from_ssat(load_ssat(ssat_path), tmp)
28
+ return tmp
29
+
30
+
31
+ def verify_tests(root: Path, feature: str, ssat_path: Path) -> GateResult:
32
+ root = Path(root)
33
+ try:
34
+ checks = _load_manifest(root, feature)
35
+ except ValueError as exc:
36
+ attr = Attribution("verify_tests", 1, 0, [
37
+ UnitResult(
38
+ unit="verify_tests:manifest",
39
+ stage="verify_tests",
40
+ passed=False,
41
+ evidence=str(exc),
42
+ failure_class=FailureClass.HOLLOW_MANIFEST,
43
+ )
44
+ ])
45
+ return GateResult(False, attr)
46
+
47
+ if checks is None:
48
+ attr = Attribution("verify_tests", 1, 0, [
49
+ UnitResult(
50
+ unit="verify_tests:manifest",
51
+ stage="verify_tests",
52
+ passed=False,
53
+ evidence=f"no smoke manifest at smoke/{feature}.json; nothing to verify",
54
+ failure_class=FailureClass.HOLLOW_MANIFEST,
55
+ )
56
+ ])
57
+ return GateResult(False, attr)
58
+
59
+ if not checks:
60
+ attr = Attribution("verify_tests", 1, 0, [
61
+ UnitResult(
62
+ unit="verify_tests:manifest",
63
+ stage="verify_tests",
64
+ passed=False,
65
+ evidence="smoke manifest declares no checks",
66
+ failure_class=FailureClass.HOLLOW_MANIFEST,
67
+ )
68
+ ])
69
+ return GateResult(False, attr)
70
+
71
+ if all(not check.must_fail_on_stub for check in checks):
72
+ units = [
73
+ UnitResult(
74
+ unit=f"verify_tests:{check.name}",
75
+ stage="verify_tests",
76
+ passed=False,
77
+ evidence="every check is exempt; manifest verifies no behavior",
78
+ failure_class=FailureClass.HOLLOW_MANIFEST,
79
+ )
80
+ for check in checks
81
+ ]
82
+ return GateResult(False, Attribution("verify_tests", len(units), 0, units))
83
+
84
+ stub_root = materialize_stub_root(ssat_path)
85
+ try:
86
+ units: list[UnitResult] = []
87
+ for check in checks:
88
+ unit = f"verify_tests:{check.name}"
89
+ if not check.must_fail_on_stub:
90
+ units.append(UnitResult(
91
+ unit=unit,
92
+ stage="verify_tests",
93
+ passed=True,
94
+ evidence="exempt: declared structural check",
95
+ failure_class=None,
96
+ ))
97
+ continue
98
+
99
+ result = _run_check(check, stub_root)
100
+ hollow = result.passed
101
+ units.append(UnitResult(
102
+ unit=unit,
103
+ stage="verify_tests",
104
+ passed=not hollow,
105
+ evidence=(
106
+ "check PASSED against an empty stub; it asserts nothing the "
107
+ "implementation provides"
108
+ if hollow
109
+ else f"correctly failed on stub: {result.reason}"
110
+ ),
111
+ failure_class=FailureClass.HOLLOW_TEST if hollow else None,
112
+ ))
113
+ finally:
114
+ shutil.rmtree(stub_root, ignore_errors=True)
115
+
116
+ attr = Attribution("verify_tests", len(units), sum(unit.passed for unit in units), units)
117
+ return GateResult(attr.rate == 1.0, attr)
118
+
@@ -0,0 +1,194 @@
1
+ """forgeline runtime smoke gate — behavior-by-inspection before ship.
2
+
3
+ The rest of ForgeLine verifies code against *specifications*: the judge checks
4
+ consistency, the QA audit grades static quality, the intent thread proves the
5
+ code honors the sealed envelope. All of that is correctness-*by-construction*.
6
+
7
+ None of it answers the question a per-PR preview deployment answers: **does the
8
+ built thing actually RUN and behave correctly when executed?** A change can pass
9
+ every static gate and still crash on import, throw at runtime, or produce the
10
+ wrong output. This gate closes that gap at the right scale for a solo builder:
11
+ it runs the artifact against declared behavioral checks and blocks ship on any
12
+ runtime failure — without the cost of ephemeral per-PR environments.
13
+
14
+ Design contract:
15
+ - Deterministic pass/fail: a check either ran green or it didn't.
16
+ - Isolated: checks run in a subprocess with a timeout, so a hang or crash in
17
+ the built code cannot take down the orchestrator.
18
+ - Declarative: behavioral checks live in a `smoke/` manifest beside the spec,
19
+ so what "correct runtime behavior" means is itself a reviewed artifact.
20
+ - Fail-closed: no manifest, or a manifest that references nothing runnable,
21
+ is a BLOCK — you cannot ship unverified runtime behavior by omission.
22
+ """
23
+ from __future__ import annotations
24
+ import json
25
+ import subprocess
26
+ import sys
27
+ import time
28
+ from dataclasses import dataclass, field
29
+ from pathlib import Path
30
+ from forgeline.attribution import Attribution, FailureClass, GateResult, UnitResult
31
+
32
+
33
+ @dataclass
34
+ class SmokeCheck:
35
+ name: str
36
+ kind: str # "command" | "python"
37
+ run: str # shell command, or python snippet/file
38
+ expect_exit: int = 0
39
+ expect_stdout: str | None = None # substring that must appear
40
+ timeout_s: int = 30
41
+ must_fail_on_stub: bool = True
42
+
43
+
44
+ @dataclass
45
+ class SmokeResult:
46
+ name: str
47
+ passed: bool
48
+ reason: str
49
+ duration_ms: int = 0
50
+
51
+
52
+ @dataclass
53
+ class SmokeReport:
54
+ results: list[SmokeResult] = field(default_factory=list)
55
+ manifest_found: bool = True
56
+
57
+ @property
58
+ def ok(self) -> bool:
59
+ return self.manifest_found and bool(self.results) and all(r.passed for r in self.results)
60
+
61
+ @property
62
+ def failures(self):
63
+ return [r for r in self.results if not r.passed]
64
+
65
+ def add(self, name, passed, reason, duration_ms=0):
66
+ self.results.append(SmokeResult(name, passed, reason, duration_ms))
67
+
68
+ @property
69
+ def attribution(self) -> Attribution:
70
+ units = []
71
+ for result in self.results:
72
+ failure_class = None
73
+ reason = result.reason.lower()
74
+ if not result.passed:
75
+ if "timed out" in reason:
76
+ failure_class = FailureClass.RUNTIME_TIMEOUT
77
+ elif "expected stdout" in reason:
78
+ failure_class = FailureClass.WRONG_OUTPUT
79
+ else:
80
+ failure_class = FailureClass.RUNTIME_CRASH
81
+ units.append(UnitResult(
82
+ unit=f"smoke:{result.name}",
83
+ stage="smoke",
84
+ passed=result.passed,
85
+ evidence=result.reason,
86
+ failure_class=failure_class,
87
+ ))
88
+ return Attribution("smoke", len(units), sum(unit.passed for unit in units), units)
89
+
90
+ @property
91
+ def gate_result(self) -> GateResult:
92
+ attr = self.attribution
93
+ return GateResult(attr.n_checked > 0 and attr.rate == 1.0, attr)
94
+
95
+
96
+ def _load_manifest(root: Path, feature: str) -> list[SmokeCheck] | None:
97
+ """A smoke manifest is JSON at smoke/<feature>.json (or smoke/smoke.json).
98
+ Each entry declares one behavioral check. Missing manifest -> None (fail-closed
99
+ upstream)."""
100
+ candidates = [
101
+ Path(root) / "smoke" / f"{feature}.json",
102
+ Path(root) / "smoke" / "smoke.json",
103
+ ]
104
+ for p in candidates:
105
+ if p.exists():
106
+ try:
107
+ data = json.loads(p.read_text())
108
+ except json.JSONDecodeError as e:
109
+ raise ValueError(f"smoke manifest {p} is not valid JSON: {e}")
110
+ checks = []
111
+ for entry in data.get("checks", []):
112
+ checks.append(SmokeCheck(
113
+ name=entry["name"],
114
+ kind=entry.get("kind", "command"),
115
+ run=entry["run"],
116
+ expect_exit=entry.get("expect_exit", 0),
117
+ expect_stdout=entry.get("expect_stdout"),
118
+ timeout_s=entry.get("timeout_s", 30),
119
+ must_fail_on_stub=entry.get("must_fail_on_stub", True),
120
+ ))
121
+ return checks
122
+ return None
123
+
124
+
125
+ def _run_check(check: SmokeCheck, cwd: Path) -> SmokeResult:
126
+ """Execute one check in an isolated subprocess with a hard timeout."""
127
+ t0 = time.monotonic()
128
+ if check.kind == "python":
129
+ cmd = [sys.executable, "-c", check.run]
130
+ else:
131
+ cmd = check.run # shell command string
132
+ try:
133
+ proc = subprocess.run(
134
+ cmd,
135
+ cwd=str(cwd),
136
+ shell=(check.kind != "python"),
137
+ capture_output=True,
138
+ text=True,
139
+ timeout=check.timeout_s,
140
+ )
141
+ except subprocess.TimeoutExpired:
142
+ ms = int((time.monotonic() - t0) * 1000)
143
+ return SmokeResult(check.name, False,
144
+ f"timed out after {check.timeout_s}s — runtime hang", ms)
145
+ except (OSError, ValueError) as e:
146
+ ms = int((time.monotonic() - t0) * 1000)
147
+ return SmokeResult(check.name, False, f"could not execute: {e}", ms)
148
+
149
+ ms = int((time.monotonic() - t0) * 1000)
150
+ # exit-code check
151
+ if proc.returncode != check.expect_exit:
152
+ tail = (proc.stderr or proc.stdout or "").strip().splitlines()[-3:]
153
+ return SmokeResult(check.name, False,
154
+ f"exit {proc.returncode} != expected {check.expect_exit}; "
155
+ f"last: {' | '.join(tail)[:200]}", ms)
156
+ # stdout-substring check
157
+ if check.expect_stdout is not None and check.expect_stdout not in (proc.stdout or ""):
158
+ return SmokeResult(check.name, False,
159
+ f"expected stdout to contain {check.expect_stdout!r}, not found", ms)
160
+ return SmokeResult(check.name, True, "ran green", ms)
161
+
162
+
163
+ def runtime_smoke(root: Path, feature: str) -> SmokeReport:
164
+ """Run all declared behavioral checks for a feature. Fail-closed: no manifest,
165
+ empty manifest, or any failing check blocks ship."""
166
+ rep = SmokeReport()
167
+ root = Path(root)
168
+ try:
169
+ checks = _load_manifest(root, feature)
170
+ except ValueError as e:
171
+ rep.manifest_found = True
172
+ rep.add("manifest", False, str(e))
173
+ return rep
174
+ if checks is None:
175
+ rep.manifest_found = False
176
+ rep.add("manifest", False,
177
+ f"no smoke manifest at smoke/{feature}.json — cannot verify runtime "
178
+ f"behavior. Declare at least one behavioral check before ship.")
179
+ return rep
180
+ if not checks:
181
+ rep.add("manifest", False, "smoke manifest present but declares no checks — "
182
+ "runtime behavior would ship unverified.")
183
+ return rep
184
+ for c in checks:
185
+ rep.results.append(_run_check(c, root))
186
+ return rep
187
+
188
+
189
+ def smoke_report_lines(rep: SmokeReport) -> list[str]:
190
+ out = []
191
+ for r in rep.results:
192
+ mark = "PASS" if r.passed else "FAIL"
193
+ out.append(f"[{mark}] {r.name} ({r.duration_ms}ms): {r.reason}")
194
+ return out
@@ -0,0 +1,14 @@
1
+ """Skill-check gate — ensures the AGENT EXPERIENCE stays high: the run
2
+ produced the receipts, the state machine advanced legally, and the skill
3
+ memory recorded a lesson. Keeps the factory itself from eroding."""
4
+ from __future__ import annotations
5
+ from pathlib import Path
6
+
7
+ def skill_check(root: Path, feature: str) -> tuple[bool, list[str]]:
8
+ findings = []
9
+ fd = Path(root)/".forge"/feature
10
+ if not (fd/"receipts.jsonl").exists() or not (fd/"receipts.jsonl").read_text().strip():
11
+ findings.append("S_NO_RECEIPTS run produced no receipts")
12
+ if not (fd/"state.json").exists():
13
+ findings.append("S_NO_STATE run has no state record")
14
+ return (len(findings) == 0, findings)