codefence 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
codefence.py ADDED
@@ -0,0 +1,4397 @@
1
+ #!/usr/bin/env python3
2
+ # Copyright (c) 2026 CodeFence. All rights reserved.
3
+ # Licensed under the CodeFence End User License Agreement.
4
+ # See LICENSE.txt for terms. Source is provided for auditability.
5
+ # Redistribution and resale are prohibited.
6
+ """
7
+ CodeFence - Offline AI Code Sanity Check
8
+ Version 1.0.0
9
+
10
+ A single-file, zero-dependency, offline CLI that scans AI-generated
11
+ code for selected dangerous patterns before commit.
12
+
13
+ This is a pattern-based sanity check. It is NOT a security audit and
14
+ does NOT replace professional security review.
15
+ """
16
+ from __future__ import annotations
17
+
18
+ import argparse
19
+ import ast
20
+ import hashlib
21
+ import html as _html
22
+ import json
23
+ import os
24
+ import re
25
+ import sqlite3
26
+ import stat
27
+ import subprocess
28
+ import sys
29
+ import tempfile
30
+ import time
31
+ from dataclasses import dataclass
32
+ from datetime import datetime, timezone
33
+ from enum import Enum
34
+ from pathlib import Path
35
+ from typing import Callable, Iterable, Iterator, Sequence
36
+
37
+
38
+ # =============================================================================
39
+ # [SECTION] Constants
40
+ # =============================================================================
41
+
42
+ TOOL_NAME = "codefence"
43
+ TOOL_VERSION = "1.0.0"
44
+ RULES_SCHEMA = "codefence/rules-v1"
45
+ DEFAULT_MAX_SIZE = 2 * 1024 * 1024
46
+ DEFAULT_RULES_FILENAME = "rules.json"
47
+
48
+ EXIT_OK = 0
49
+ EXIT_FINDINGS = 1
50
+ EXIT_USAGE = 2
51
+ EXIT_INTERNAL = 3
52
+
53
+ # Security limits (see SECURITY.md)
54
+ MAX_REGEX_PATTERN_LEN = 500
55
+ MAX_REGEX_PATTERNS_PER_RULE = 20
56
+ MAX_REGEX_LINE_LEN = 8192
57
+ REGEX_TIMEOUT_SEC = 0.5
58
+ NO_SIGNAL_MAX_REGEX_INPUT = 1024 # strict cap when signal is unavailable
59
+ _NO_SIGNAL_WARNED = False
60
+
61
+ _DEBUG_HANDLERS = False # set True to see handler exceptions
62
+
63
+
64
+ # =============================================================================
65
+ # [SECTION] Enums
66
+ # =============================================================================
67
+
68
+ class Severity(str, Enum):
69
+ CRITICAL = "critical"
70
+ HIGH = "high"
71
+ MEDIUM = "medium"
72
+ LOW = "low"
73
+ INFO = "info"
74
+
75
+
76
+ _SEVERITY_RANK: dict[Severity, int] = {
77
+ Severity.CRITICAL: 0,
78
+ Severity.HIGH: 1,
79
+ Severity.MEDIUM: 2,
80
+ Severity.LOW: 3,
81
+ Severity.INFO: 4,
82
+ }
83
+
84
+
85
+ def severity_rank(value: Severity) -> int:
86
+ return _SEVERITY_RANK[value]
87
+
88
+
89
+ class Language(str, Enum):
90
+ PYTHON = "python"
91
+ JAVASCRIPT = "javascript"
92
+
93
+
94
+ class Detection(str, Enum):
95
+ REGEX = "regex"
96
+ AST = "ast"
97
+ LEXICAL = "lexical"
98
+
99
+
100
+ class Confidence(str, Enum):
101
+ HIGH = "high"
102
+ MEDIUM = "medium"
103
+ LOW = "low"
104
+
105
+
106
+ # =============================================================================
107
+ # [SECTION] Data model (frozen, immutable, hashable)
108
+ # =============================================================================
109
+
110
+ class TokenKind(str, Enum):
111
+ IDENT = "ident"
112
+ KEYWORD = "keyword"
113
+ STRING = "string"
114
+ TEMPLATE = "template"
115
+ NUMBER = "number"
116
+ REGEX = "regex"
117
+ PUNCT = "punct"
118
+
119
+
120
+ @dataclass(frozen=True, slots=True)
121
+ class Token:
122
+ kind: TokenKind
123
+ value: str
124
+ line: int
125
+ column: int
126
+
127
+
128
+ @dataclass(frozen=True, slots=True)
129
+ class Finding:
130
+ id: str
131
+ rule_name: str
132
+ severity: Severity
133
+ category: str
134
+ file: str
135
+ line: int
136
+ column: int
137
+ snippet: str
138
+ message: str
139
+ remediation: str
140
+ language: Language
141
+ confidence: Confidence
142
+ fix_before: str = ""
143
+ fix_after: str = ""
144
+ action: str = "block"
145
+
146
+ def sort_key(self) -> tuple[str, int, int, str]:
147
+ return (self.file, self.line, self.column, self.id)
148
+
149
+
150
+ @dataclass(frozen=True, slots=True)
151
+ class Rule:
152
+ id: str
153
+ name: str
154
+ languages: tuple[Language, ...]
155
+ severity: Severity
156
+ category: str
157
+ description: str
158
+ detection: Detection
159
+ message: str
160
+ remediation: str
161
+ references: tuple[str, ...]
162
+ confidence: Confidence
163
+ enabled: bool
164
+ patterns: tuple[re.Pattern[str], ...] = ()
165
+ handler: str = ""
166
+ fix_before: str = ""
167
+ fix_after: str = ""
168
+ action: str = "block"
169
+ rule_version: str = "1"
170
+ fingerprint: str = ""
171
+
172
+
173
+ @dataclass(frozen=True, slots=True)
174
+ class ScanContext:
175
+ path: str
176
+ language: Language
177
+ source: str
178
+ lines: tuple[str, ...]
179
+ ast_tree: ast.AST | None = None
180
+ js_tokens: tuple[Token, ...] = ()
181
+
182
+
183
+ # =============================================================================
184
+ # [SECTION] Rule handler registry (populated incrementally)
185
+ # =============================================================================
186
+
187
+ RuleHandler = Callable[[ScanContext, Rule], Iterable[Finding]]
188
+
189
+ RULE_HANDLERS: dict[str, RuleHandler] = {}
190
+
191
+
192
+ def register_handler(name: str) -> Callable[[RuleHandler], RuleHandler]:
193
+ """Decorator to register an AST/lexical rule handler by name."""
194
+ def deco(fn: RuleHandler) -> RuleHandler:
195
+ RULE_HANDLERS[name] = fn
196
+ return fn
197
+ return deco
198
+
199
+
200
+ # =============================================================================
201
+ # [SECTION] Rule loading (pure)
202
+ # =============================================================================
203
+
204
+ def load_rules(path: Path) -> list[Rule]:
205
+ raw = json.loads(path.read_text(encoding="utf-8"))
206
+ schema = raw.get("schema")
207
+ if schema != RULES_SCHEMA:
208
+ raise ValueError(f"Unsupported rules schema: {schema!r}")
209
+ entries = raw.get("rules", [])
210
+ if not isinstance(entries, list):
211
+ raise ValueError("rules.json: 'rules' must be a list")
212
+ return [_build_rule(e) for e in entries]
213
+
214
+
215
+ VALID_ACTIONS = frozenset({"allow", "warn", "block"})
216
+
217
+
218
+ def _validate_action(value, rid: str) -> str:
219
+ if value is None:
220
+ return "block"
221
+ if not isinstance(value, str):
222
+ raise ValueError(f"rule {rid}: action must be a string")
223
+ low = value.lower()
224
+ if low not in VALID_ACTIONS:
225
+ raise ValueError(
226
+ f"rule {rid}: action must be one of {sorted(VALID_ACTIONS)}, "
227
+ f"got {value!r}"
228
+ )
229
+ return low
230
+
231
+
232
+ def _build_rule(entry: dict) -> Rule:
233
+ if not isinstance(entry, dict):
234
+ raise ValueError(f"rule entry must be an object, got {type(entry).__name__}")
235
+ rid = entry.get("id")
236
+ if not isinstance(rid, str) or not re.match(r"^(R[0-9]{3}|ORG-[0-9]{3,})$", rid):
237
+ raise ValueError(
238
+ f"rule id must match R### or ORG-###, got {rid!r}"
239
+ )
240
+ detection_str = entry.get("detection")
241
+ try:
242
+ detection = Detection(detection_str)
243
+ except ValueError:
244
+ raise ValueError(f"rule {rid}: invalid detection {detection_str!r}")
245
+
246
+ patterns: tuple[re.Pattern[str], ...] = ()
247
+ if detection == Detection.REGEX:
248
+ raw_patterns = entry.get("patterns", [])
249
+ if not isinstance(raw_patterns, list):
250
+ raise ValueError(f"rule {rid}: patterns must be a list")
251
+ if len(raw_patterns) > MAX_REGEX_PATTERNS_PER_RULE:
252
+ raise ValueError(
253
+ f"rule {rid}: too many patterns "
254
+ f"({len(raw_patterns)} > {MAX_REGEX_PATTERNS_PER_RULE})"
255
+ )
256
+ compiled: list[re.Pattern[str]] = []
257
+ for i, pat in enumerate(raw_patterns):
258
+ if not isinstance(pat, str):
259
+ raise ValueError(f"rule {rid}: pattern {i} must be a string")
260
+ if len(pat) > MAX_REGEX_PATTERN_LEN:
261
+ raise ValueError(
262
+ f"rule {rid}: pattern {i} too long "
263
+ f"({len(pat)} > {MAX_REGEX_PATTERN_LEN})"
264
+ )
265
+ try:
266
+ compiled.append(re.compile(pat))
267
+ except re.error as e:
268
+ raise ValueError(f"rule {rid}: pattern {i} invalid regex: {e}")
269
+ patterns = tuple(compiled)
270
+
271
+ try:
272
+ languages = tuple(Language(x) for x in entry["languages"])
273
+ except (KeyError, TypeError, ValueError) as e:
274
+ raise ValueError(f"rule {rid}: invalid languages: {e}")
275
+ try:
276
+ severity = Severity(entry["severity"])
277
+ except (KeyError, ValueError) as e:
278
+ raise ValueError(f"rule {rid}: invalid severity: {e}")
279
+ try:
280
+ confidence = Confidence(entry.get("confidence", "medium"))
281
+ except ValueError as e:
282
+ raise ValueError(f"rule {rid}: invalid confidence: {e}")
283
+
284
+ return Rule(
285
+ id=rid,
286
+ name=str(entry.get("name", "")),
287
+ languages=languages,
288
+ severity=severity,
289
+ category=str(entry.get("category", "")),
290
+ description=str(entry.get("description", "")),
291
+ detection=detection,
292
+ message=str(entry.get("message", "")),
293
+ remediation=str(entry.get("remediation", "")),
294
+ references=tuple(str(x) for x in entry.get("references", [])),
295
+ confidence=confidence,
296
+ enabled=bool(entry.get("enabled", True)),
297
+ patterns=patterns,
298
+ handler=str(entry.get("handler", "") or ""),
299
+ fix_before=str(entry.get("fix_before", "")),
300
+ fix_after=str(entry.get("fix_after", "")),
301
+ action=_validate_action(entry.get("action", "block"), rid),
302
+ rule_version=str(entry.get("rule_version", "1")),
303
+ fingerprint=str(entry.get("fingerprint", "")),
304
+ )
305
+
306
+
307
+ # =============================================================================
308
+ # [SECTION] Python AST helpers
309
+ # =============================================================================
310
+
311
+ def _node_line(node: ast.AST) -> int:
312
+ return int(getattr(node, "lineno", 0) or 0)
313
+
314
+
315
+ def _node_col(node: ast.AST) -> int:
316
+ return int(getattr(node, "col_offset", 0) or 0) + 1
317
+
318
+
319
+ def _make_finding(ctx: ScanContext, rule: Rule, node: ast.AST,
320
+ message: str | None = None) -> Finding:
321
+ line = _node_line(node)
322
+ if 1 <= line <= len(ctx.lines):
323
+ snippet = ctx.lines[line - 1].strip()
324
+ else:
325
+ snippet = ""
326
+ if len(snippet) > 200:
327
+ snippet = snippet[:197] + "..."
328
+ return Finding(
329
+ id=rule.id,
330
+ rule_name=rule.name,
331
+ severity=rule.severity,
332
+ category=rule.category,
333
+ file=ctx.path,
334
+ line=line,
335
+ column=_node_col(node),
336
+ snippet=snippet,
337
+ message=message or rule.message,
338
+ remediation=rule.remediation,
339
+ language=ctx.language,
340
+ confidence=rule.confidence,
341
+ fix_before=rule.fix_before,
342
+ fix_after=rule.fix_after,
343
+ action=rule.action,
344
+ )
345
+
346
+
347
+ def _walk(tree: ast.AST) -> Iterable[ast.AST]:
348
+ yield from ast.walk(tree)
349
+
350
+
351
+ def _attr_name(node: ast.AST) -> str | None:
352
+ if isinstance(node, ast.Name):
353
+ return node.id
354
+ if isinstance(node, ast.Attribute):
355
+ base = _attr_name(node.value)
356
+ return f"{base}.{node.attr}" if base else node.attr
357
+ return None
358
+
359
+
360
+ def _is_string_constant(node: ast.AST) -> bool:
361
+ return isinstance(node, ast.Constant) and isinstance(node.value, str)
362
+
363
+
364
+ SECURITY_TOKEN_NAMES = (
365
+ "token", "secret", "session", "password", "passwd", "pwd",
366
+ "key", "nonce", "otp", "salt", "csrf", "auth",
367
+ )
368
+
369
+
370
+ def _looks_security_related(name: str) -> bool:
371
+ low = name.lower()
372
+ return any(tok in low for tok in SECURITY_TOKEN_NAMES)
373
+
374
+
375
+ # =============================================================================
376
+ # [SECTION] Python rule handlers
377
+ # =============================================================================
378
+
379
+ @register_handler("check_sql_injection_python")
380
+ def check_sql_injection_python(ctx: ScanContext, rule: Rule) -> Iterable[Finding]:
381
+ if ctx.ast_tree is None:
382
+ return
383
+ execute_methods = {"execute", "executemany", "executescript"}
384
+ for node in _walk(ctx.ast_tree):
385
+ if not isinstance(node, ast.Call):
386
+ continue
387
+ if not isinstance(node.func, ast.Attribute):
388
+ continue
389
+ if node.func.attr not in execute_methods:
390
+ continue
391
+ if not node.args:
392
+ continue
393
+ if _is_dangerous_sql(node.args[0]):
394
+ yield _make_finding(ctx, rule, node)
395
+
396
+
397
+ def _is_dangerous_sql(node: ast.AST) -> bool:
398
+ if isinstance(node, ast.JoinedStr):
399
+ return True
400
+ if isinstance(node, ast.BinOp):
401
+ if isinstance(node.op, ast.Add):
402
+ return _is_string_constant(node.left) or _is_string_constant(node.right)
403
+ if isinstance(node.op, ast.Mod):
404
+ return _is_string_constant(node.left)
405
+ if isinstance(node, ast.Call):
406
+ if isinstance(node.func, ast.Attribute) and node.func.attr == "format":
407
+ return _is_string_constant(node.func.value)
408
+ return False
409
+
410
+
411
+ @register_handler("check_eval_exec_python")
412
+ def check_eval_exec_python(ctx: ScanContext, rule: Rule) -> Iterable[Finding]:
413
+ if ctx.ast_tree is None:
414
+ return
415
+ dangerous = {"eval", "exec"}
416
+ for node in _walk(ctx.ast_tree):
417
+ if not isinstance(node, ast.Call):
418
+ continue
419
+ if not isinstance(node.func, ast.Name):
420
+ continue
421
+ if node.func.id not in dangerous:
422
+ continue
423
+ if not node.args:
424
+ continue
425
+ first = node.args[0]
426
+ if isinstance(first, ast.Constant) and isinstance(first.value, (int, float)):
427
+ continue
428
+ yield _make_finding(ctx, rule, node)
429
+
430
+
431
+ @register_handler("check_shell_true_python")
432
+ def check_shell_true_python(ctx: ScanContext, rule: Rule) -> Iterable[Finding]:
433
+ if ctx.ast_tree is None:
434
+ return
435
+ shell_funcs = {
436
+ "subprocess.run", "subprocess.call", "subprocess.Popen",
437
+ "subprocess.check_call", "subprocess.check_output",
438
+ }
439
+ always_shell = {"os.system", "os.popen"}
440
+ for node in _walk(ctx.ast_tree):
441
+ if not isinstance(node, ast.Call):
442
+ continue
443
+ name = _attr_name(node.func)
444
+ if name in shell_funcs:
445
+ for kw in node.keywords:
446
+ if kw.arg == "shell" and isinstance(kw.value, ast.Constant) \
447
+ and kw.value.value is True:
448
+ yield _make_finding(ctx, rule, node)
449
+ break
450
+ elif name in always_shell:
451
+ yield _make_finding(ctx, rule, node)
452
+
453
+
454
+ @register_handler("check_random_python")
455
+ def check_random_python(ctx: ScanContext, rule: Rule) -> Iterable[Finding]:
456
+ if ctx.ast_tree is None:
457
+ return
458
+ random_attrs = {
459
+ "random", "randint", "randrange", "choice", "choices",
460
+ "sample", "uniform", "getrandbits", "shuffle",
461
+ }
462
+ for node in _walk(ctx.ast_tree):
463
+ if not isinstance(node, ast.Assign):
464
+ continue
465
+ value = node.value
466
+ if not (isinstance(value, ast.Call)
467
+ and isinstance(value.func, ast.Attribute)
468
+ and value.func.attr in random_attrs
469
+ and isinstance(value.func.value, ast.Name)
470
+ and value.func.value.id == "random"):
471
+ continue
472
+ names: list[str] = []
473
+ for tgt in node.targets:
474
+ if isinstance(tgt, ast.Name):
475
+ names.append(tgt.id)
476
+ elif isinstance(tgt, ast.Attribute):
477
+ names.append(tgt.attr)
478
+ if any(_looks_security_related(n) for n in names):
479
+ yield _make_finding(ctx, rule, node)
480
+
481
+
482
+ @register_handler("check_except_pass_python")
483
+ def check_except_pass_python(ctx: ScanContext, rule: Rule) -> Iterable[Finding]:
484
+ if ctx.ast_tree is None:
485
+ return
486
+ for node in _walk(ctx.ast_tree):
487
+ if not isinstance(node, ast.Try):
488
+ continue
489
+ for handler in node.handlers:
490
+ if not _is_broad_except(handler.type):
491
+ continue
492
+ if _is_silent_body(handler.body):
493
+ yield _make_finding(ctx, rule, handler)
494
+
495
+
496
+ def _is_broad_except(exc_type: ast.AST | None) -> bool:
497
+ if exc_type is None:
498
+ return True
499
+ if isinstance(exc_type, ast.Name):
500
+ return exc_type.id in {"Exception", "BaseException"}
501
+ if isinstance(exc_type, ast.Tuple):
502
+ return any(
503
+ isinstance(e, ast.Name) and e.id in {"Exception", "BaseException"}
504
+ for e in exc_type.elts
505
+ )
506
+ return False
507
+
508
+
509
+ def _is_silent_body(body: list[ast.stmt]) -> bool:
510
+ if not body:
511
+ return False
512
+ meaningful = [s for s in body
513
+ if not (isinstance(s, ast.Expr)
514
+ and isinstance(s.value, ast.Constant)
515
+ and isinstance(s.value.value, str))]
516
+ if len(meaningful) != 1:
517
+ return False
518
+ return isinstance(meaningful[0], ast.Pass)
519
+
520
+
521
+ # =============================================================================
522
+ # [SECTION] Python additional rule handlers (Day 4)
523
+ # =============================================================================
524
+
525
+ MUTABLE_DEFAULT_TYPES = (ast.List, ast.Dict, ast.Set)
526
+ MUTABLE_DEFAULT_CALLS = {"list", "dict", "set", "bytearray", "OrderedDict", "defaultdict"}
527
+
528
+
529
+ @register_handler("check_mutable_defaults_python")
530
+ def check_mutable_defaults_python(ctx: ScanContext, rule: Rule) -> Iterable[Finding]:
531
+ if ctx.ast_tree is None:
532
+ return
533
+ for node in _walk(ctx.ast_tree):
534
+ if not isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef)):
535
+ continue
536
+ defaults = list(node.args.defaults)
537
+ defaults.extend(d for d in node.args.kw_defaults if d is not None)
538
+ for default in defaults:
539
+ if _is_mutable_default(default):
540
+ yield _make_finding(
541
+ ctx, rule, node,
542
+ f"Function '{node.name}' has a mutable default argument.",
543
+ )
544
+ break
545
+
546
+
547
+ def _is_mutable_default(node: ast.AST) -> bool:
548
+ if isinstance(node, MUTABLE_DEFAULT_TYPES):
549
+ return True
550
+ if isinstance(node, ast.Call) and isinstance(node.func, ast.Name):
551
+ if node.func.id in MUTABLE_DEFAULT_CALLS:
552
+ return True
553
+ return False
554
+
555
+
556
+ @register_handler("check_unused_imports_python")
557
+ def check_unused_imports_python(ctx: ScanContext, rule: Rule) -> Iterable[Finding]:
558
+ if ctx.ast_tree is None:
559
+ return
560
+ tree = ctx.ast_tree
561
+ imported: dict[str, ast.AST] = {}
562
+ for node in ast.walk(tree):
563
+ if isinstance(node, ast.Import):
564
+ for alias in node.names:
565
+ if alias.asname:
566
+ key = alias.asname
567
+ elif "." in alias.name:
568
+ key = alias.name.split(".")[0]
569
+ else:
570
+ key = alias.name
571
+ imported.setdefault(key, node)
572
+ elif isinstance(node, ast.ImportFrom):
573
+ # Skip compiler directives (from __future__ import ...)
574
+ if node.module == "__future__":
575
+ continue
576
+ for alias in node.names:
577
+ if alias.name == "*":
578
+ continue
579
+ key = alias.asname or alias.name
580
+ imported.setdefault(key, node)
581
+ if not imported:
582
+ return
583
+ used: set[str] = set()
584
+ for node in ast.walk(tree):
585
+ if isinstance(node, ast.Name):
586
+ used.add(node.id)
587
+ elif isinstance(node, ast.Attribute):
588
+ root: ast.AST = node
589
+ while isinstance(root, ast.Attribute):
590
+ root = root.value
591
+ if isinstance(root, ast.Name):
592
+ used.add(root.id)
593
+ for name, node in imported.items():
594
+ if name not in used:
595
+ yield _make_finding(ctx, rule, node, f"Unused import: {name}")
596
+
597
+
598
+ ROUTE_METHODS = {"get", "post", "put", "delete", "patch", "route", "options", "head"}
599
+
600
+ AUTH_DECORATOR_TOKENS = (
601
+ "login_required", "requires_auth", "auth_required", "jwt_required",
602
+ "requires_login", "authenticated", "protected", "token_required",
603
+ "requires_authentication", "requires_token",
604
+ )
605
+
606
+ ADMIN_PATH_HINTS = ("admin", "manage", "internal", "private", "superuser")
607
+
608
+ LOGIN_PATH_HINTS = ("login", "signin", "register", "signup", "forgot", "reset", "auth", "token", "otp")
609
+
610
+
611
+ def _decorator_call_name(deco: ast.AST) -> str:
612
+ target = deco.func if isinstance(deco, ast.Call) else deco
613
+ return _attr_name(target) or ""
614
+
615
+
616
+ @register_handler("check_missing_auth")
617
+ def check_missing_auth(ctx: ScanContext, rule: Rule) -> Iterable[Finding]:
618
+ if ctx.language == Language.JAVASCRIPT:
619
+ yield from check_missing_auth_js(ctx, rule)
620
+ return
621
+ if ctx.ast_tree is None:
622
+ return
623
+ for node in _walk(ctx.ast_tree):
624
+ if not isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef)):
625
+ continue
626
+ route_path: str | None = None
627
+ has_auth = False
628
+ for deco in node.decorator_list:
629
+ name = _decorator_call_name(deco)
630
+ last = name.split(".")[-1] if name else ""
631
+ if last in ROUTE_METHODS and isinstance(deco, ast.Call) \
632
+ and deco.args and _is_string_constant(deco.args[0]):
633
+ route_path = deco.args[0].value # type: ignore[union-attr]
634
+ if any(tok in name.lower() for tok in AUTH_DECORATOR_TOKENS):
635
+ has_auth = True
636
+ if route_path is None or has_auth:
637
+ continue
638
+ if any(h in route_path.lower() for h in ADMIN_PATH_HINTS):
639
+ yield _make_finding(
640
+ ctx, rule, node,
641
+ f"Route '{route_path}' has no visible authentication.",
642
+ )
643
+
644
+
645
+ @register_handler("check_missing_rate_limit")
646
+ def check_missing_rate_limit(ctx: ScanContext, rule: Rule) -> Iterable[Finding]:
647
+ if ctx.language == Language.JAVASCRIPT:
648
+ yield from check_missing_rate_limit_js(ctx, rule)
649
+ return
650
+ if ctx.ast_tree is None:
651
+ return
652
+ for node in _walk(ctx.ast_tree):
653
+ if not isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef)):
654
+ continue
655
+ route_path: str | None = None
656
+ has_limiter = False
657
+ for deco in node.decorator_list:
658
+ name = _decorator_call_name(deco)
659
+ last = name.split(".")[-1] if name else ""
660
+ if last in ROUTE_METHODS and isinstance(deco, ast.Call) \
661
+ and deco.args and _is_string_constant(deco.args[0]):
662
+ route_path = deco.args[0].value # type: ignore[union-attr]
663
+ if any(tok in name.lower() for tok in ("limit", "throttle", "ratelimit")):
664
+ has_limiter = True
665
+ if route_path is None or has_limiter:
666
+ continue
667
+ if any(t in route_path.lower() for t in LOGIN_PATH_HINTS):
668
+ yield _make_finding(
669
+ ctx, rule, node,
670
+ f"Endpoint '{route_path}' has no visible rate limiting.",
671
+ )
672
+
673
+
674
+ @register_handler("check_open_redirect")
675
+ def check_open_redirect(ctx: ScanContext, rule: Rule) -> Iterable[Finding]:
676
+ if ctx.language == Language.JAVASCRIPT:
677
+ yield from check_open_redirect_js(ctx, rule)
678
+ return
679
+ if ctx.ast_tree is None:
680
+ return
681
+ for node in _walk(ctx.ast_tree):
682
+ if not isinstance(node, ast.Call):
683
+ continue
684
+ name = _attr_name(node.func) or ""
685
+ if not name.endswith("redirect"):
686
+ continue
687
+ if not node.args:
688
+ continue
689
+ arg = node.args[0]
690
+ if isinstance(arg, ast.Constant) and isinstance(arg.value, str):
691
+ continue
692
+ if _is_user_input_access(arg):
693
+ yield _make_finding(ctx, rule, node)
694
+
695
+
696
+ def _is_user_input_access(node: ast.AST) -> bool:
697
+ if isinstance(node, ast.Call):
698
+ name = _attr_name(node.func) or ""
699
+ low = name.lower()
700
+ if name.startswith("request."):
701
+ return True
702
+ if "query" in low or "param" in low or "get_param" in low:
703
+ return True
704
+ if isinstance(node, ast.Subscript):
705
+ name = _attr_name(node.value) or ""
706
+ if name.startswith("request."):
707
+ return True
708
+ return False
709
+
710
+
711
+ @register_handler("check_jwt_no_exp")
712
+ def check_jwt_no_exp(ctx: ScanContext, rule: Rule) -> Iterable[Finding]:
713
+ if ctx.language == Language.JAVASCRIPT:
714
+ yield from check_jwt_no_exp_js(ctx, rule)
715
+ return
716
+ if ctx.ast_tree is None:
717
+ return
718
+ for node in _walk(ctx.ast_tree):
719
+ if not isinstance(node, ast.Call):
720
+ continue
721
+ name = _attr_name(node.func) or ""
722
+ if not name.endswith("jwt.encode"):
723
+ continue
724
+ if not node.args:
725
+ continue
726
+ payload = node.args[0]
727
+ if not isinstance(payload, ast.Dict):
728
+ continue
729
+ keys = [
730
+ k.value for k in payload.keys
731
+ if isinstance(k, ast.Constant) and isinstance(k.value, str)
732
+ ]
733
+ if "exp" not in keys and "expires" not in keys and "iat" not in keys:
734
+ yield _make_finding(ctx, rule, node)
735
+
736
+
737
+ @register_handler("check_race_conditions")
738
+ def check_race_conditions(ctx: ScanContext, rule: Rule) -> Iterable[Finding]:
739
+ if ctx.language != Language.PYTHON:
740
+ return
741
+ if ctx.ast_tree is None:
742
+ return
743
+ for fn in _walk(ctx.ast_tree):
744
+ if not isinstance(fn, (ast.FunctionDef, ast.AsyncFunctionDef)):
745
+ continue
746
+ for i, stmt in enumerate(fn.body):
747
+ if not isinstance(stmt, ast.If):
748
+ continue
749
+ if not _cond_is_exists_check(stmt.test):
750
+ continue
751
+ is_negated = _cond_is_negated(stmt.test)
752
+ if is_negated:
753
+ # Pattern 1: if not exists(x): open(x, "w") ...
754
+ for sub in ast.walk(stmt):
755
+ if isinstance(sub, ast.Call) and _is_write_open(sub):
756
+ yield _make_finding(ctx, rule, sub)
757
+ break
758
+ else:
759
+ # Pattern 2: if exists(x): return / raise / continue
760
+ # open(x, "w") # after the guard
761
+ if not _body_is_early_exit(stmt.body):
762
+ continue
763
+ for later in fn.body[i + 1:]:
764
+ fire = None
765
+ for sub in ast.walk(later):
766
+ if isinstance(sub, ast.Call) and _is_write_open(sub):
767
+ fire = sub
768
+ break
769
+ if fire is not None:
770
+ yield _make_finding(ctx, rule, fire)
771
+ break
772
+
773
+
774
+ def _cond_is_negated(test: ast.AST) -> bool:
775
+ return isinstance(test, ast.UnaryOp) and isinstance(test.op, ast.Not)
776
+
777
+
778
+ def _body_is_early_exit(body: list) -> bool:
779
+ if not body:
780
+ return False
781
+ if len(body) == 1 and isinstance(body[0], (ast.Return, ast.Raise,
782
+ ast.Continue, ast.Break)):
783
+ return True
784
+ return False
785
+
786
+
787
+ def _cond_is_exists_check(test: ast.AST) -> bool:
788
+ for n in ast.walk(test):
789
+ if isinstance(n, ast.Call):
790
+ name = _attr_name(n.func) or ""
791
+ if name.endswith("exists") or name.endswith("isfile") or name.endswith("isdir"):
792
+ return True
793
+ return False
794
+
795
+
796
+ def _is_write_open(node: ast.Call) -> bool:
797
+ name = _attr_name(node.func) or ""
798
+ if name not in ("open", "io.open"):
799
+ return False
800
+ if len(node.args) >= 2:
801
+ mode = node.args[1]
802
+ if isinstance(mode, ast.Constant) and isinstance(mode.value, str):
803
+ return any(c in mode.value for c in ("w", "a", "x", "+"))
804
+ for kw in node.keywords:
805
+ if kw.arg == "mode" and isinstance(kw.value, ast.Constant):
806
+ return any(c in str(kw.value.value) for c in ("w", "a", "x", "+"))
807
+ return False
808
+
809
+
810
+ @register_handler("check_path_traversal")
811
+ def check_path_traversal(ctx: ScanContext, rule: Rule) -> Iterable[Finding]:
812
+ if ctx.language != Language.PYTHON:
813
+ return
814
+ if ctx.ast_tree is None:
815
+ return
816
+ for node in _walk(ctx.ast_tree):
817
+ if not isinstance(node, ast.Call):
818
+ continue
819
+ name = _attr_name(node.func) or ""
820
+ if name not in ("open", "io.open"):
821
+ continue
822
+ if not node.args:
823
+ continue
824
+ if _looks_path_traversable(node.args[0]):
825
+ yield _make_finding(ctx, rule, node)
826
+
827
+
828
+ def _looks_path_traversable(arg: ast.AST) -> bool:
829
+ if isinstance(arg, ast.Call):
830
+ name = _attr_name(arg.func) or ""
831
+ if name.endswith("path.join"):
832
+ for a in arg.args[1:]:
833
+ if isinstance(a, (ast.Name, ast.Attribute, ast.Subscript)):
834
+ return True
835
+ if isinstance(arg, ast.BinOp) and isinstance(arg.op, ast.Add):
836
+ if isinstance(arg.left, (ast.Name, ast.Attribute, ast.Subscript)):
837
+ return True
838
+ if isinstance(arg.right, (ast.Name, ast.Attribute, ast.Subscript)):
839
+ return True
840
+ return False
841
+
842
+
843
+ # =============================================================================
844
+ # [SECTION] JavaScript lexer
845
+ # =============================================================================
846
+
847
+ JS_KEYWORDS = frozenset({
848
+ "break", "case", "catch", "class", "const", "continue", "debugger",
849
+ "default", "delete", "do", "else", "export", "extends", "finally",
850
+ "for", "function", "if", "import", "in", "instanceof", "let", "new",
851
+ "return", "super", "switch", "this", "throw", "try", "typeof", "var",
852
+ "void", "while", "with", "yield", "async", "await", "of", "static",
853
+ "get", "set",
854
+ })
855
+
856
+ JS_PUNCT_3 = ("===", "!==", "**=", "<<=", ">>=", ">>>", "&&=", "||=", "??=", "...")
857
+ JS_PUNCT_2 = ("=>", "==", "!=", "<=", ">=", "&&", "||", "??", "?.",
858
+ "++", "--", "+=", "-=", "*=", "/=", "%=", "&=", "|=", "^=",
859
+ "**", "<<", ">>")
860
+ JS_PUNCT_1 = tuple("+-*/%=<>!&|^~?:;,.(){}[]")
861
+
862
+
863
+ def tokenize_js(source: str) -> list[Token]:
864
+ """Minimal JS lexer. Comments skipped. Template literals treated as a
865
+ single token (contents not analyzed). Regex vs division resolved via
866
+ previous-token heuristic."""
867
+ tokens: list[Token] = []
868
+ i = 0
869
+ line = 1
870
+ col = 1
871
+ n = len(source)
872
+
873
+ def advance(k: int = 1) -> None:
874
+ nonlocal i, line, col
875
+ for _ in range(k):
876
+ if i >= n:
877
+ return
878
+ if source[i] == "\n":
879
+ line += 1
880
+ col = 1
881
+ else:
882
+ col += 1
883
+ i += 1
884
+
885
+ def prev_meaningful() -> Token | None:
886
+ return tokens[-1] if tokens else None
887
+
888
+ def regex_context() -> bool:
889
+ prev = prev_meaningful()
890
+ if prev is None:
891
+ return True
892
+ if prev.kind in (TokenKind.IDENT, TokenKind.NUMBER, TokenKind.STRING,
893
+ TokenKind.TEMPLATE, TokenKind.REGEX):
894
+ return False
895
+ if prev.kind == TokenKind.KEYWORD:
896
+ return prev.value not in ("this", "super", "true", "false",
897
+ "null", "undefined")
898
+ if prev.kind == TokenKind.PUNCT:
899
+ return prev.value not in (")", "]", "}")
900
+ return True
901
+
902
+ while i < n:
903
+ c = source[i]
904
+
905
+ if c in " \t\r\n":
906
+ advance()
907
+ continue
908
+
909
+ if c == "/" and i + 1 < n and source[i + 1] == "/":
910
+ while i < n and source[i] != "\n":
911
+ advance()
912
+ continue
913
+
914
+ if c == "/" and i + 1 < n and source[i + 1] == "*":
915
+ advance(2)
916
+ while i + 1 < n and not (source[i] == "*" and source[i + 1] == "/"):
917
+ advance()
918
+ if i + 1 < n:
919
+ advance(2)
920
+ continue
921
+
922
+ start_line, start_col = line, col
923
+
924
+ if c in ("'", '"'):
925
+ quote = c
926
+ buf = [c]
927
+ advance()
928
+ while i < n and source[i] != quote:
929
+ if source[i] == "\\" and i + 1 < n:
930
+ buf.append(source[i]); advance()
931
+ buf.append(source[i]); advance()
932
+ elif source[i] == "\n":
933
+ break
934
+ else:
935
+ buf.append(source[i]); advance()
936
+ if i < n and source[i] == quote:
937
+ buf.append(quote); advance()
938
+ tokens.append(Token(TokenKind.STRING, "".join(buf),
939
+ start_line, start_col))
940
+ continue
941
+
942
+ if c == "`":
943
+ buf = [c]
944
+ advance()
945
+ while i < n and source[i] != "`":
946
+ if source[i] == "\\" and i + 1 < n:
947
+ buf.append(source[i]); advance()
948
+ buf.append(source[i]); advance()
949
+ else:
950
+ buf.append(source[i]); advance()
951
+ if i < n and source[i] == "`":
952
+ buf.append("`"); advance()
953
+ tokens.append(Token(TokenKind.TEMPLATE, "".join(buf),
954
+ start_line, start_col))
955
+ continue
956
+
957
+ if c.isalpha() or c == "_" or c == "$":
958
+ buf = [c]; advance()
959
+ while i < n and (source[i].isalnum() or source[i] in "_$"):
960
+ buf.append(source[i]); advance()
961
+ word = "".join(buf)
962
+ kind = TokenKind.KEYWORD if word in JS_KEYWORDS else TokenKind.IDENT
963
+ tokens.append(Token(kind, word, start_line, start_col))
964
+ continue
965
+
966
+ if c.isdigit() or (c == "." and i + 1 < n and source[i + 1].isdigit()):
967
+ buf = [c]; advance()
968
+ while i < n and (source[i].isalnum() or source[i] in "._"):
969
+ buf.append(source[i]); advance()
970
+ tokens.append(Token(TokenKind.NUMBER, "".join(buf),
971
+ start_line, start_col))
972
+ continue
973
+
974
+ if c == "/" and regex_context():
975
+ buf = [c]; advance()
976
+ while i < n and source[i] != "/":
977
+ if source[i] == "\\" and i + 1 < n:
978
+ buf.append(source[i]); advance()
979
+ buf.append(source[i]); advance()
980
+ elif source[i] == "\n":
981
+ break
982
+ else:
983
+ buf.append(source[i]); advance()
984
+ if i < n and source[i] == "/":
985
+ buf.append("/"); advance()
986
+ while i < n and source[i].isalpha():
987
+ buf.append(source[i]); advance()
988
+ tokens.append(Token(TokenKind.REGEX, "".join(buf),
989
+ start_line, start_col))
990
+ continue
991
+
992
+ matched = False
993
+ for length, table in ((3, JS_PUNCT_3), (2, JS_PUNCT_2), (1, JS_PUNCT_1)):
994
+ chunk = source[i:i + length]
995
+ if len(chunk) == length and chunk in table:
996
+ tokens.append(Token(TokenKind.PUNCT, chunk,
997
+ start_line, start_col))
998
+ advance(length)
999
+ matched = True
1000
+ break
1001
+ if matched:
1002
+ continue
1003
+
1004
+ advance()
1005
+
1006
+ return tokens
1007
+
1008
+
1009
+ # =============================================================================
1010
+ # [SECTION] JavaScript rule handlers (Day 6)
1011
+ # =============================================================================
1012
+
1013
+ def _js_finding(ctx: ScanContext, rule: Rule, token: Token,
1014
+ message: str | None = None) -> Finding:
1015
+ line = token.line
1016
+ if 1 <= line <= len(ctx.lines):
1017
+ snippet = ctx.lines[line - 1].strip()
1018
+ else:
1019
+ snippet = ""
1020
+ if len(snippet) > 200:
1021
+ snippet = snippet[:197] + "..."
1022
+ return Finding(
1023
+ id=rule.id,
1024
+ rule_name=rule.name,
1025
+ severity=rule.severity,
1026
+ category=rule.category,
1027
+ file=ctx.path,
1028
+ line=line,
1029
+ column=token.column,
1030
+ snippet=snippet,
1031
+ message=message or rule.message,
1032
+ remediation=rule.remediation,
1033
+ language=ctx.language,
1034
+ confidence=rule.confidence,
1035
+ fix_before=rule.fix_before,
1036
+ fix_after=rule.fix_after,
1037
+ action=rule.action,
1038
+ )
1039
+
1040
+
1041
+ def _js_next(toks: tuple[Token, ...], i: int, offset: int) -> Token | None:
1042
+ j = i + offset
1043
+ if 0 <= j < len(toks):
1044
+ return toks[j]
1045
+ return None
1046
+
1047
+
1048
+ @register_handler("check_innerhtml_js")
1049
+ def check_innerhtml_js(ctx: ScanContext, rule: Rule) -> Iterable[Finding]:
1050
+ if ctx.language != Language.JAVASCRIPT:
1051
+ return
1052
+ toks = ctx.js_tokens
1053
+ for i, t in enumerate(toks):
1054
+ if t.kind == TokenKind.IDENT and t.value == "innerHTML":
1055
+ n1 = _js_next(toks, i, 1)
1056
+ n2 = _js_next(toks, i, 2)
1057
+ if (n1 and n2 and n1.kind == TokenKind.PUNCT and n1.value == "="
1058
+ and n2.kind != TokenKind.STRING):
1059
+ yield _js_finding(ctx, rule, t)
1060
+ if t.kind == TokenKind.IDENT and t.value == "document":
1061
+ n1 = _js_next(toks, i, 1)
1062
+ n2 = _js_next(toks, i, 2)
1063
+ if (n1 and n2 and n1.kind == TokenKind.PUNCT and n1.value == "."
1064
+ and n2.kind == TokenKind.IDENT and n2.value == "write"):
1065
+ yield _js_finding(ctx, rule, t)
1066
+
1067
+
1068
+ @register_handler("check_prototype_pollution_js")
1069
+ def check_prototype_pollution_js(ctx: ScanContext, rule: Rule) -> Iterable[Finding]:
1070
+ if ctx.language != Language.JAVASCRIPT:
1071
+ return
1072
+ for t in ctx.js_tokens:
1073
+ if t.kind == TokenKind.IDENT and t.value == "__proto__":
1074
+ yield _js_finding(ctx, rule, t)
1075
+
1076
+
1077
+ @register_handler("check_missing_helmet_js")
1078
+ def check_missing_helmet_js(ctx: ScanContext, rule: Rule) -> Iterable[Finding]:
1079
+ if ctx.language != Language.JAVASCRIPT:
1080
+ return
1081
+ toks = ctx.js_tokens
1082
+ has_express = False
1083
+ has_helmet = False
1084
+ express_tok: Token | None = None
1085
+ for i, t in enumerate(toks):
1086
+ if t.kind != TokenKind.STRING:
1087
+ continue
1088
+ if t.value in ("'express'", '"express"'):
1089
+ has_express = True
1090
+ if express_tok is None:
1091
+ express_tok = t
1092
+ if t.value in ("'helmet'", '"helmet"'):
1093
+ has_helmet = True
1094
+ if has_express and not has_helmet and express_tok is not None:
1095
+ yield _js_finding(ctx, rule, express_tok)
1096
+
1097
+
1098
+ @register_handler("check_promise_rejection_js")
1099
+ def check_promise_rejection_js(ctx: ScanContext, rule: Rule) -> Iterable[Finding]:
1100
+ if ctx.language != Language.JAVASCRIPT:
1101
+ return
1102
+ toks = ctx.js_tokens
1103
+ n = len(toks)
1104
+ for i, t in enumerate(toks):
1105
+ if t.kind != TokenKind.PUNCT or t.value != ".":
1106
+ continue
1107
+ nxt = _js_next(toks, i, 1)
1108
+ if not (nxt and nxt.kind == TokenKind.IDENT and nxt.value == "then"):
1109
+ continue
1110
+ # Look ahead until ';' or EOF for a .catch on the same chain
1111
+ has_catch = False
1112
+ j = i + 1
1113
+ depth = 0
1114
+ while j < n:
1115
+ tk = toks[j]
1116
+ if tk.kind == TokenKind.PUNCT:
1117
+ if tk.value in ("(", "[", "{"):
1118
+ depth += 1
1119
+ elif tk.value in (")", "]", "}"):
1120
+ if depth > 0:
1121
+ depth -= 1
1122
+ elif tk.value == ";" and depth == 0:
1123
+ break
1124
+ elif tk.value == "." and depth == 0:
1125
+ nx = _js_next(toks, j, 1)
1126
+ if nx and nx.kind in (TokenKind.IDENT, TokenKind.KEYWORD) and nx.value == "catch":
1127
+ has_catch = True
1128
+ break
1129
+ j += 1
1130
+ if not has_catch:
1131
+ yield _js_finding(ctx, rule, nxt,
1132
+ "Promise .then() without .catch() in the same chain.")
1133
+
1134
+
1135
+ # =============================================================================
1136
+ # [SECTION] JS extensions to shared heuristic rules
1137
+ # =============================================================================
1138
+
1139
+ JS_AUTH_TOKENS = (
1140
+ "auth", "authenticate", "authorize", "requireAuth", "ensureAuth",
1141
+ "isAuthenticated", "verifyToken", "passport",
1142
+ )
1143
+
1144
+ JS_RATE_TOKENS = (
1145
+ "rateLimit", "rateLimiter", "limiter", "throttle", "slowDown",
1146
+ "expressRateLimit",
1147
+ )
1148
+
1149
+ JS_LOGIN_HINTS = ("login", "signin", "register", "signup", "forgot",
1150
+ "reset", "auth", "token", "otp")
1151
+
1152
+ JS_ADMIN_HINTS = ("admin", "manage", "internal", "private", "superuser")
1153
+
1154
+
1155
+ def _js_route_targets(toks: tuple[Token, ...]) -> list[tuple[Token, str]]:
1156
+ """Return (route_token, path) for patterns:
1157
+ app.get('/path', ...), router.post('/x', ...), app.route('/y')"""
1158
+ routes: list[tuple[Token, str]] = []
1159
+ methods = {"get", "post", "put", "delete", "patch", "options", "head", "route"}
1160
+ for i, t in enumerate(toks):
1161
+ if t.kind != TokenKind.PUNCT or t.value != ".":
1162
+ continue
1163
+ n1 = _js_next(toks, i, 1)
1164
+ n2 = _js_next(toks, i, 2)
1165
+ n3 = _js_next(toks, i, 3)
1166
+ if not (n1 and n2 and n3):
1167
+ continue
1168
+ if n1.kind not in (TokenKind.IDENT, TokenKind.KEYWORD) or n1.value not in methods:
1169
+ continue
1170
+ if n2.kind != TokenKind.PUNCT or n2.value != "(":
1171
+ continue
1172
+ if n3.kind != TokenKind.STRING:
1173
+ continue
1174
+ raw = n3.value
1175
+ if raw[:1] in ("'", '"'):
1176
+ raw = raw[1:]
1177
+ if raw[-1:] in ("'", '"'):
1178
+ raw = raw[:-1]
1179
+ if not raw.startswith("/"):
1180
+ continue
1181
+ routes.append((n3, raw))
1182
+ return routes
1183
+
1184
+
1185
+ def _js_has_nearby_auth(toks: tuple[Token, ...], start_idx: int,
1186
+ tokens_of_interest: tuple[str, ...],
1187
+ span: int = 200) -> bool:
1188
+ end = min(len(toks), start_idx + span)
1189
+ for k in range(start_idx, end):
1190
+ tk = toks[k]
1191
+ if tk.kind in (TokenKind.IDENT, TokenKind.STRING):
1192
+ low = tk.value.lower()
1193
+ if any(tok.lower() in low for tok in tokens_of_interest):
1194
+ return True
1195
+ return False
1196
+
1197
+
1198
+ def _find_token_index(toks: tuple[Token, ...], target: Token) -> int:
1199
+ for i, tk in enumerate(toks):
1200
+ if tk is target:
1201
+ return i
1202
+ return 0
1203
+
1204
+
1205
+ def check_missing_auth_js(ctx: ScanContext, rule: Rule) -> Iterable[Finding]:
1206
+ if ctx.language != Language.JAVASCRIPT:
1207
+ return
1208
+ toks = ctx.js_tokens
1209
+ for tok, path in _js_route_targets(toks):
1210
+ if not any(h in path.lower() for h in JS_ADMIN_HINTS):
1211
+ continue
1212
+ idx = _find_token_index(toks, tok)
1213
+ if not _js_has_nearby_auth(toks, idx, JS_AUTH_TOKENS):
1214
+ yield _js_finding(ctx, rule, tok,
1215
+ f"Route '{path}' has no visible authentication.")
1216
+
1217
+
1218
+ def check_missing_rate_limit_js(ctx: ScanContext, rule: Rule) -> Iterable[Finding]:
1219
+ if ctx.language != Language.JAVASCRIPT:
1220
+ return
1221
+ toks = ctx.js_tokens
1222
+ for tok, path in _js_route_targets(toks):
1223
+ if not any(h in path.lower() for h in JS_LOGIN_HINTS):
1224
+ continue
1225
+ idx = _find_token_index(toks, tok)
1226
+ if not _js_has_nearby_auth(toks, idx, JS_RATE_TOKENS):
1227
+ yield _js_finding(ctx, rule, tok,
1228
+ f"Endpoint '{path}' has no visible rate limiting.")
1229
+
1230
+
1231
+ def check_open_redirect_js(ctx: ScanContext, rule: Rule) -> Iterable[Finding]:
1232
+ if ctx.language != Language.JAVASCRIPT:
1233
+ return
1234
+ toks = ctx.js_tokens
1235
+ for i, t in enumerate(toks):
1236
+ if t.kind != TokenKind.PUNCT or t.value != ".":
1237
+ continue
1238
+ n1 = _js_next(toks, i, 1)
1239
+ if not (n1 and n1.kind == TokenKind.IDENT and n1.value == "redirect"):
1240
+ continue
1241
+ n2 = _js_next(toks, i, 2)
1242
+ n3 = _js_next(toks, i, 3)
1243
+ if not (n2 and n3 and n2.value == "("):
1244
+ continue
1245
+ if n3.kind == TokenKind.STRING:
1246
+ continue
1247
+ # check if user-controlled (req.query / req.params)
1248
+ seg = toks[i:i + 12]
1249
+ joined = "".join(tk.value for tk in seg)
1250
+ if "req.query" in joined or "req.params" in joined or "req.body" in joined:
1251
+ yield _js_finding(ctx, rule, n1)
1252
+
1253
+
1254
+ def check_jwt_no_exp_js(ctx: ScanContext, rule: Rule) -> Iterable[Finding]:
1255
+ if ctx.language != Language.JAVASCRIPT:
1256
+ return
1257
+ toks = ctx.js_tokens
1258
+ for i, t in enumerate(toks):
1259
+ if t.kind != TokenKind.IDENT:
1260
+ continue
1261
+ if t.value not in ("sign", "encode"):
1262
+ continue
1263
+ if t.kind not in (TokenKind.IDENT, TokenKind.KEYWORD):
1264
+ continue
1265
+ prev = _js_next(toks, i, -1)
1266
+ if not (prev and prev.kind == TokenKind.PUNCT and prev.value == "."):
1267
+ continue
1268
+ # Only flag if there is a jwt-ish receiver within previous 4 tokens
1269
+ back = toks[max(0, i - 4):i]
1270
+ if not any(tk.kind == TokenKind.IDENT and "jwt" in tk.value.lower()
1271
+ for tk in back):
1272
+ continue
1273
+ # Look ahead to matching ) and check for expiresIn / exp
1274
+ j = i + 1
1275
+ depth = 0
1276
+ has_exp = False
1277
+ while j < len(toks):
1278
+ tk = toks[j]
1279
+ if tk.kind == TokenKind.PUNCT:
1280
+ if tk.value == "(":
1281
+ depth += 1
1282
+ elif tk.value == ")":
1283
+ if depth == 0:
1284
+ break
1285
+ depth -= 1
1286
+ if tk.kind == TokenKind.IDENT and tk.value in ("expiresIn", "exp"):
1287
+ has_exp = True
1288
+ break
1289
+ j += 1
1290
+ if not has_exp:
1291
+ yield _js_finding(ctx, rule, t,
1292
+ "JWT signed without expiration (expiresIn/exp).")
1293
+
1294
+
1295
+ # =============================================================================
1296
+ # [SECTION] Python additional rule handlers (Day 7)
1297
+ # =============================================================================
1298
+
1299
+ REQUESTS_METHODS = frozenset({
1300
+ "get", "post", "put", "delete", "patch", "head", "options", "request",
1301
+ })
1302
+
1303
+
1304
+ @register_handler("check_requests_no_timeout_python")
1305
+ def check_requests_no_timeout_python(ctx: ScanContext, rule: Rule) -> Iterable[Finding]:
1306
+ if ctx.language != Language.PYTHON or ctx.ast_tree is None:
1307
+ return
1308
+ for node in _walk(ctx.ast_tree):
1309
+ if not isinstance(node, ast.Call):
1310
+ continue
1311
+ if not isinstance(node.func, ast.Attribute):
1312
+ continue
1313
+ if node.func.attr not in REQUESTS_METHODS:
1314
+ continue
1315
+ base = node.func.value
1316
+ if not (isinstance(base, ast.Name) and base.id == "requests"):
1317
+ continue
1318
+ has_timeout = any(kw.arg == "timeout" for kw in node.keywords)
1319
+ if not has_timeout:
1320
+ yield _make_finding(ctx, rule, node)
1321
+
1322
+
1323
+ DATETIME_METHODS = frozenset({
1324
+ "now", "utcnow", "fromtimestamp", "fromordinal",
1325
+ })
1326
+
1327
+
1328
+ @register_handler("check_datetime_no_tz_python")
1329
+ def check_datetime_no_tz_python(ctx: ScanContext, rule: Rule) -> Iterable[Finding]:
1330
+ if ctx.language != Language.PYTHON or ctx.ast_tree is None:
1331
+ return
1332
+ for node in _walk(ctx.ast_tree):
1333
+ if not isinstance(node, ast.Call):
1334
+ continue
1335
+ if not isinstance(node.func, ast.Attribute):
1336
+ continue
1337
+ if node.func.attr not in DATETIME_METHODS:
1338
+ continue
1339
+ base = node.func.value
1340
+ is_datetime = False
1341
+ if isinstance(base, ast.Attribute) and base.attr == "datetime":
1342
+ if isinstance(base.value, ast.Name) and base.value.id == "datetime":
1343
+ is_datetime = True
1344
+ elif isinstance(base, ast.Name) and base.id == "datetime":
1345
+ is_datetime = True
1346
+ if not is_datetime:
1347
+ continue
1348
+ if node.func.attr == "utcnow":
1349
+ yield _make_finding(
1350
+ ctx, rule, node,
1351
+ "datetime.utcnow() returns a naive datetime (deprecated in 3.12).",
1352
+ )
1353
+ continue
1354
+ if node.func.attr == "fromordinal":
1355
+ yield _make_finding(ctx, rule, node)
1356
+ continue
1357
+ has_tz_kw = any(kw.arg == "tz" for kw in node.keywords)
1358
+ positional_count = len(node.args)
1359
+ if node.func.attr == "fromtimestamp":
1360
+ has_tz_pos = positional_count >= 2
1361
+ else: # now
1362
+ has_tz_pos = positional_count >= 1
1363
+ if has_tz_kw or has_tz_pos:
1364
+ continue
1365
+ # Context check: only fire when the result is likely security-related.
1366
+ if _datetime_call_in_security_context(node, ctx.ast_tree):
1367
+ yield _make_finding(ctx, rule, node)
1368
+
1369
+
1370
+ DATETIME_SECURITY_TOKENS = (
1371
+ "exp", "expire", "expiry", "expires", "issued", "issue",
1372
+ "iat", "token", "session", "auth", "jwt", "nonce", "salt",
1373
+ "otp", "csrf", "reset", "sign",
1374
+ )
1375
+
1376
+ DATETIME_SECURITY_FUNCS = (
1377
+ "login", "register", "authenticate", "signin", "signup",
1378
+ "create_token", "issue_token", "make_token", "generate_token",
1379
+ "new_session", "create_session", "reset_password",
1380
+ )
1381
+
1382
+
1383
+ def _datetime_call_in_security_context(node: ast.AST,
1384
+ tree: ast.AST | None) -> bool:
1385
+ """Return True if the datetime call's result is assigned to a
1386
+ security-related name, or if the call is inside a security-named
1387
+ function. Parents are found via a reverse walk over the tree."""
1388
+ if tree is None:
1389
+ return False
1390
+ parents: dict[int, ast.AST] = {}
1391
+ for parent in ast.walk(tree):
1392
+ for child in ast.iter_child_nodes(parent):
1393
+ parents[id(child)] = parent
1394
+
1395
+ target_names: list[str] = []
1396
+ cursor: ast.AST = node
1397
+ security_func_found = False
1398
+ for _ in range(20): # bounded depth
1399
+ parent = parents.get(id(cursor))
1400
+ if parent is None:
1401
+ break
1402
+ if isinstance(parent, (ast.FunctionDef, ast.AsyncFunctionDef)):
1403
+ fname = parent.name.lower()
1404
+ if any(tok in fname for tok in DATETIME_SECURITY_FUNCS):
1405
+ security_func_found = True
1406
+ break # stop at function boundary
1407
+ if isinstance(parent, ast.Assign):
1408
+ for tgt in parent.targets:
1409
+ if isinstance(tgt, ast.Name):
1410
+ target_names.append(tgt.id)
1411
+ elif isinstance(tgt, ast.Attribute):
1412
+ target_names.append(tgt.attr)
1413
+ if isinstance(parent, ast.AnnAssign) and isinstance(parent.target,
1414
+ ast.Name):
1415
+ target_names.append(parent.target.id)
1416
+ if isinstance(parent, ast.keyword):
1417
+ # e.g. jwt.encode(..., exp=datetime.now()) -> kw.arg == "exp"
1418
+ if parent.arg:
1419
+ target_names.append(parent.arg)
1420
+ cursor = parent
1421
+
1422
+ if security_func_found:
1423
+ return True
1424
+ for name in target_names:
1425
+ low = name.lower()
1426
+ if any(tok in low for tok in DATETIME_SECURITY_TOKENS):
1427
+ return True
1428
+ return False
1429
+
1430
+
1431
+ @register_handler("check_env_fallback_secret")
1432
+ def check_env_fallback_secret(ctx: ScanContext, rule: Rule) -> Iterable[Finding]:
1433
+ if ctx.language == Language.PYTHON:
1434
+ yield from _check_env_fallback_python(ctx, rule)
1435
+ elif ctx.language == Language.JAVASCRIPT:
1436
+ yield from _check_env_fallback_js(ctx, rule)
1437
+
1438
+
1439
+ ENV_GET_FUNCS = frozenset({
1440
+ "getenv", "get", "os.environ.get", "environ.get",
1441
+ })
1442
+
1443
+
1444
+ def _name_looks_secret(name: str) -> bool:
1445
+ low = name.lower()
1446
+ return any(tok in low for tok in (
1447
+ "secret", "key", "token", "password", "passwd", "pwd",
1448
+ "apikey", "api_key", "auth", "session", "salt", "nonce",
1449
+ "csrf", "jwt",
1450
+ ))
1451
+
1452
+
1453
+ def _string_literal(node: ast.AST) -> str | None:
1454
+ if isinstance(node, ast.Constant) and isinstance(node.value, str):
1455
+ return node.value
1456
+ return None
1457
+
1458
+
1459
+ def _check_env_fallback_python(ctx: ScanContext,
1460
+ rule: Rule) -> Iterable[Finding]:
1461
+ if ctx.ast_tree is None:
1462
+ return
1463
+ for node in _walk(ctx.ast_tree):
1464
+ if not isinstance(node, ast.Call):
1465
+ continue
1466
+ name = _attr_name(node.func) or ""
1467
+ if name not in ("os.environ.get", "os.getenv", "environ.get"):
1468
+ continue
1469
+ if len(node.args) < 2:
1470
+ continue
1471
+ env_name = _string_literal(node.args[0])
1472
+ if env_name is None or not _name_looks_secret(env_name):
1473
+ continue
1474
+ fallback = _string_literal(node.args[1])
1475
+ if fallback is None:
1476
+ continue
1477
+ if fallback.strip() == "" or fallback.strip().lower() in (
1478
+ "none", "null", "undefined",
1479
+ ):
1480
+ continue
1481
+ yield _make_finding(
1482
+ ctx, rule, node,
1483
+ f"Env var '{env_name}' has a hardcoded fallback secret.",
1484
+ )
1485
+
1486
+
1487
+ def _check_env_fallback_js(ctx: ScanContext,
1488
+ rule: Rule) -> Iterable[Finding]:
1489
+ toks = ctx.js_tokens
1490
+ n = len(toks)
1491
+ for i, t in enumerate(toks):
1492
+ if t.kind != TokenKind.IDENT or t.value != "process":
1493
+ continue
1494
+ n1 = _js_next(toks, i, 1)
1495
+ n2 = _js_next(toks, i, 2)
1496
+ n3 = _js_next(toks, i, 3)
1497
+ if not (n1 and n2 and n3):
1498
+ continue
1499
+ if not (n1.kind == TokenKind.PUNCT and n1.value == "."
1500
+ and n2.kind == TokenKind.IDENT and n2.value == "env"
1501
+ and n3.kind == TokenKind.PUNCT and n3.value == "."):
1502
+ continue
1503
+ env_tok = _js_next(toks, i, 4)
1504
+ if not (env_tok and env_tok.kind == TokenKind.IDENT):
1505
+ continue
1506
+ if not _name_looks_secret(env_tok.value):
1507
+ continue
1508
+ # Look ahead for || "fallback" or ?? "fallback"
1509
+ j = i + 5
1510
+ while j < n:
1511
+ tk = toks[j]
1512
+ if tk.kind == TokenKind.PUNCT and tk.value in ("||", "??"):
1513
+ fallback = _js_next(toks, j, 1)
1514
+ if fallback and fallback.kind == TokenKind.STRING:
1515
+ inner = fallback.value.strip("\"'")
1516
+ if inner.strip():
1517
+ yield _js_finding(
1518
+ ctx, rule, env_tok,
1519
+ f"Env var '{env_tok.value}' has a hardcoded "
1520
+ f"fallback secret.",
1521
+ )
1522
+ break
1523
+ if tk.kind == TokenKind.PUNCT and tk.value in (";", ",", ")"):
1524
+ break
1525
+ j += 1
1526
+
1527
+
1528
+ # =============================================================================
1529
+ # [SECTION] Secure cache (opt-in, off by default)
1530
+ # =============================================================================
1531
+
1532
+ CACHE_DIR_NAME = "codefence"
1533
+ CACHE_VERSION = "v1"
1534
+
1535
+
1536
+ def _cache_root() -> Path:
1537
+ base = os.environ.get("XDG_CACHE_HOME")
1538
+ if base:
1539
+ root = Path(base) / CACHE_DIR_NAME
1540
+ else:
1541
+ root = Path.home() / ".cache" / CACHE_DIR_NAME
1542
+ return root / CACHE_VERSION
1543
+
1544
+
1545
+ def _ensure_cache_dir() -> Path:
1546
+ root = _cache_root()
1547
+ root.mkdir(parents=True, exist_ok=True, mode=0o700)
1548
+ try:
1549
+ os.chmod(root, 0o700)
1550
+ except OSError:
1551
+ pass
1552
+ return root
1553
+
1554
+
1555
+ def _cache_key(source: str, language: Language,
1556
+ rules_fingerprint: str) -> str:
1557
+ h = hashlib.sha256()
1558
+ h.update(CACHE_VERSION.encode("ascii"))
1559
+ h.update(b"\x00")
1560
+ h.update(language.value.encode("ascii"))
1561
+ h.update(b"\x00")
1562
+ h.update(rules_fingerprint.encode("ascii"))
1563
+ h.update(b"\x00")
1564
+ h.update(source.encode("utf-8", errors="replace"))
1565
+ return h.hexdigest()
1566
+
1567
+
1568
+ def _rules_fingerprint(rules_path: Path) -> str:
1569
+ try:
1570
+ data = rules_path.read_bytes()
1571
+ except OSError:
1572
+ return "0" * 64
1573
+ return hashlib.sha256(data).hexdigest()
1574
+
1575
+
1576
+ def _finding_to_dict(f: Finding) -> dict:
1577
+ return {
1578
+ "id": f.id,
1579
+ "rule_name": f.rule_name,
1580
+ "severity": f.severity.value,
1581
+ "category": f.category,
1582
+ "file": f.file,
1583
+ "line": f.line,
1584
+ "column": f.column,
1585
+ "snippet": f.snippet,
1586
+ "message": f.message,
1587
+ "remediation": f.remediation,
1588
+ "language": f.language.value,
1589
+ "confidence": f.confidence.value,
1590
+ }
1591
+
1592
+
1593
+ def _dict_to_finding(d: dict) -> Finding | None:
1594
+ try:
1595
+ return Finding(
1596
+ id=str(d["id"]),
1597
+ rule_name=str(d["rule_name"]),
1598
+ severity=Severity(d["severity"]),
1599
+ category=str(d["category"]),
1600
+ file=str(d["file"]),
1601
+ line=int(d["line"]),
1602
+ column=int(d["column"]),
1603
+ snippet=str(d["snippet"]),
1604
+ message=str(d["message"]),
1605
+ remediation=str(d["remediation"]),
1606
+ language=Language(d["language"]),
1607
+ confidence=Confidence(d["confidence"]),
1608
+ )
1609
+ except (KeyError, TypeError, ValueError):
1610
+ return None
1611
+
1612
+
1613
+ def cache_load(source: str, language: Language, rules_fp: str) -> list[Finding] | None:
1614
+ """Return cached findings, or None on miss/corruption."""
1615
+ try:
1616
+ key = _cache_key(source, language, rules_fp)
1617
+ f = _cache_root() / f"{key}.json"
1618
+ if not f.is_file():
1619
+ return None
1620
+ # Reject symlinks and non-regular files
1621
+ st = f.lstat()
1622
+ if stat.S_ISLNK(st.st_mode) or not stat.S_ISREG(st.st_mode):
1623
+ return None
1624
+ data = json.loads(f.read_text(encoding="utf-8"))
1625
+ if not isinstance(data, dict):
1626
+ return None
1627
+ if data.get("version") != CACHE_VERSION:
1628
+ return None
1629
+ items = data.get("findings", [])
1630
+ if not isinstance(items, list):
1631
+ return None
1632
+ out: list[Finding] = []
1633
+ for item in items:
1634
+ fnd = _dict_to_finding(item)
1635
+ if fnd is None:
1636
+ return None
1637
+ out.append(fnd)
1638
+ return out
1639
+ except (OSError, ValueError, json.JSONDecodeError):
1640
+ return None
1641
+
1642
+
1643
+ def cache_store(source: str, language: Language, rules_fp: str,
1644
+ findings: Sequence[Finding]) -> None:
1645
+ try:
1646
+ root = _ensure_cache_dir()
1647
+ key = _cache_key(source, language, rules_fp)
1648
+ target = root / f"{key}.json"
1649
+ payload = {
1650
+ "version": CACHE_VERSION,
1651
+ "language": language.value,
1652
+ "rules_fingerprint": rules_fp,
1653
+ "findings": [_finding_to_dict(f) for f in findings],
1654
+ }
1655
+ text = json.dumps(payload, ensure_ascii=False)
1656
+ fd, tmp_path = tempfile.mkstemp(
1657
+ prefix=".tmp-", dir=str(root), suffix=".json"
1658
+ )
1659
+ try:
1660
+ os.fchmod(fd, 0o600)
1661
+ with os.fdopen(fd, "w", encoding="utf-8") as fp:
1662
+ fp.write(text)
1663
+ os.replace(tmp_path, target)
1664
+ except Exception:
1665
+ try:
1666
+ os.unlink(tmp_path)
1667
+ except OSError:
1668
+ pass
1669
+ raise
1670
+ except (OSError, ValueError):
1671
+ # Cache is best-effort; never fail the scan because of cache
1672
+ return
1673
+
1674
+
1675
+ # =============================================================================
1676
+ # [SECTION] Source helpers
1677
+ # =============================================================================
1678
+
1679
+ def _try_parse_python(source: str) -> tuple[ast.AST | None, str | None]:
1680
+ """Return (tree, error_message). Never raises."""
1681
+ try:
1682
+ return ast.parse(source), None
1683
+ except SyntaxError as e:
1684
+ where = f"line {e.lineno}, column {e.offset}" if e.lineno else "unknown"
1685
+ return None, f"Python syntax error at {where}: {e.msg}"
1686
+ except ValueError as e:
1687
+ return None, f"Python parse error: {e}"
1688
+
1689
+
1690
+ def _parse_python(source: str) -> ast.AST | None:
1691
+ tree, _err = _try_parse_python(source)
1692
+ return tree
1693
+
1694
+
1695
+ def _make_meta_finding(path: str, language: Language, message: str) -> Finding:
1696
+ """Synthetic finding used when a file could not be scanned at all."""
1697
+ return Finding(
1698
+ id="R000",
1699
+ rule_name="File not scanned",
1700
+ severity=Severity.INFO,
1701
+ category="meta",
1702
+ file=path,
1703
+ line=1,
1704
+ column=1,
1705
+ snippet="",
1706
+ message=message,
1707
+ remediation="Fix the issue above and re-run the scanner.",
1708
+ language=language,
1709
+ confidence=Confidence.HIGH,
1710
+ )
1711
+
1712
+
1713
+ def _line_col(source: str, offset: int) -> tuple[int, int]:
1714
+ line = source.count("\n", 0, offset) + 1
1715
+ last_nl = source.rfind("\n", 0, offset)
1716
+ column = offset - last_nl
1717
+ return line, column
1718
+
1719
+
1720
+ def _snippet(source: str, start: int, end: int, radius: int = 60) -> str:
1721
+ a = max(0, start - radius)
1722
+ b = min(len(source), end + radius)
1723
+ text = source[a:b].replace("\n", " ").replace("\r", " ").replace("\t", " ")
1724
+ return " ".join(text.split())
1725
+
1726
+
1727
+ # =============================================================================
1728
+ # [SECTION] Core scan API (pure, testable)
1729
+ # =============================================================================
1730
+
1731
+ _NOQA_RE = re.compile(
1732
+ r"(?:#|//)\s*noqa(?:\s*:\s*([A-Za-z0-9_,\s-]+))?",
1733
+ re.IGNORECASE,
1734
+ )
1735
+
1736
+
1737
+ def _parse_noqa_directives(lines: Sequence[str]) -> dict[int, set[str] | None]:
1738
+ """Return {line_no: set_of_rule_ids or None}.
1739
+
1740
+ None means "suppress all rules on this line."
1741
+ A set of rule IDs means "suppress only these rules on this line."
1742
+ """
1743
+ directives: dict[int, set[str] | None] = {}
1744
+ for i, line in enumerate(lines, 1):
1745
+ m = _NOQA_RE.search(line)
1746
+ if not m:
1747
+ continue
1748
+ rules_part = m.group(1)
1749
+ if not rules_part:
1750
+ directives[i] = None
1751
+ continue
1752
+ ids = {tok.strip().upper()
1753
+ for tok in re.split(r"[,\s]+", rules_part)
1754
+ if tok.strip()}
1755
+ if not ids:
1756
+ directives[i] = None
1757
+ else:
1758
+ directives[i] = ids
1759
+ return directives
1760
+
1761
+
1762
+ def _apply_noqa(findings: Iterable[Finding],
1763
+ directives: dict[int, set[str] | None]) -> list[Finding]:
1764
+ out: list[Finding] = []
1765
+ for f in findings:
1766
+ d = directives.get(f.line, "missing")
1767
+ if d == "missing":
1768
+ out.append(f)
1769
+ continue
1770
+ if d is None:
1771
+ continue
1772
+ if f.id.upper() in d:
1773
+ continue
1774
+ out.append(f)
1775
+ return out
1776
+
1777
+
1778
+ def _dedupe_findings(findings: Iterable[Finding]) -> list[Finding]:
1779
+ """One finding per (rule_id, file, line) - drops duplicate matches
1780
+ produced when multiple patterns of the same rule hit the same line."""
1781
+ seen: set[tuple[str, str, int]] = set()
1782
+ out: list[Finding] = []
1783
+ for f in sorted(findings, key=Finding.sort_key):
1784
+ key = (f.id, f.file, f.line)
1785
+ if key in seen:
1786
+ continue
1787
+ seen.add(key)
1788
+ out.append(f)
1789
+ return out
1790
+
1791
+
1792
+ def scan_source(
1793
+ source: str,
1794
+ language: Language,
1795
+ rules: Sequence[Rule],
1796
+ path: str = "<memory>",
1797
+ ) -> list[Finding]:
1798
+ # Defensive BOM strip for direct API callers.
1799
+ if source.startswith("\ufeff"):
1800
+ source = source[1:]
1801
+ js_tokens: tuple[Token, ...] = ()
1802
+ ast_tree: ast.AST | None = None
1803
+ parse_error: str | None = None
1804
+ if language == Language.PYTHON:
1805
+ ast_tree, parse_error = _try_parse_python(source)
1806
+ elif language == Language.JAVASCRIPT:
1807
+ js_tokens = tuple(tokenize_js(source))
1808
+ ctx = ScanContext(
1809
+ path=path,
1810
+ language=language,
1811
+ source=source,
1812
+ lines=tuple(source.splitlines()),
1813
+ ast_tree=ast_tree,
1814
+ js_tokens=js_tokens,
1815
+ )
1816
+ findings: list[Finding] = []
1817
+ if parse_error is not None:
1818
+ findings.append(_make_meta_finding(path, language, parse_error))
1819
+ return findings
1820
+ for rule in rules:
1821
+ if not rule.enabled:
1822
+ continue
1823
+ if language not in rule.languages:
1824
+ continue
1825
+ if rule.detection == Detection.REGEX:
1826
+ findings.extend(_run_regex_rule(rule, ctx))
1827
+ elif rule.detection in (Detection.AST, Detection.LEXICAL):
1828
+ handler = RULE_HANDLERS.get(rule.handler)
1829
+ if handler is None:
1830
+ continue
1831
+ try:
1832
+ findings.extend(handler(ctx, rule))
1833
+ except Exception as _e:
1834
+ if _DEBUG_HANDLERS:
1835
+ import traceback
1836
+ print(f"[handler-error] {rule.id} "
1837
+ f"{rule.handler}: {_e!r}", file=sys.stderr)
1838
+ traceback.print_exc()
1839
+ continue
1840
+ directives = _parse_noqa_directives(ctx.lines)
1841
+ if directives:
1842
+ findings = _apply_noqa(findings, directives)
1843
+ return _dedupe_findings(findings)
1844
+
1845
+
1846
+ class _RegexTimeout(Exception):
1847
+ pass
1848
+
1849
+
1850
+ def _regex_timeout_handler(signum, frame):
1851
+ raise _RegexTimeout()
1852
+
1853
+
1854
+ def _safe_regex_matches(pattern: "re.Pattern[str]", text: str) -> list:
1855
+ """Run pattern.finditer(text) with a hard timeout when possible.
1856
+
1857
+ On platforms or threads where signal-based timeouts are unavailable
1858
+ (Windows, non-main threads), fall back to a strict input cap and
1859
+ emit a one-time warning. The regex still runs, but on at most
1860
+ NO_SIGNAL_MAX_REGEX_INPUT characters.
1861
+ """
1862
+ global _NO_SIGNAL_WARNED
1863
+ try:
1864
+ import signal as _signal
1865
+ prev = _signal.signal(_signal.SIGALRM, _regex_timeout_handler)
1866
+ _signal.setitimer(_signal.ITIMER_REAL, REGEX_TIMEOUT_SEC)
1867
+ except (ImportError, ValueError, AttributeError, OSError):
1868
+ # Signal-based timeout unavailable. Apply strict input cap.
1869
+ if not _NO_SIGNAL_WARNED:
1870
+ print(
1871
+ "[codefence] warning: signal-based regex timeout is not "
1872
+ "available on this platform or thread; using strict input "
1873
+ "cap instead. Regex results may be less complete on very "
1874
+ "long lines.",
1875
+ file=sys.stderr,
1876
+ )
1877
+ _NO_SIGNAL_WARNED = True
1878
+ capped = text[:NO_SIGNAL_MAX_REGEX_INPUT]
1879
+ return list(pattern.finditer(capped))
1880
+ try:
1881
+ return list(pattern.finditer(text))
1882
+ except _RegexTimeout:
1883
+ return []
1884
+ finally:
1885
+ try:
1886
+ _signal.setitimer(_signal.ITIMER_REAL, 0)
1887
+ _signal.signal(_signal.SIGALRM, prev)
1888
+ except Exception: # noqa: R012
1889
+ pass
1890
+
1891
+
1892
+ def _run_regex_rule(rule: Rule, ctx: ScanContext) -> Iterator[Finding]:
1893
+ """Scan line-by-line so a single regex call can never touch more than
1894
+ one line of source (bounds catastrophic backtracking) and enforce a
1895
+ hard per-call timeout on top of that."""
1896
+ for pattern in rule.patterns:
1897
+ for line_no, raw_line in enumerate(ctx.lines, 1):
1898
+ line = raw_line if len(raw_line) <= MAX_REGEX_LINE_LEN \
1899
+ else raw_line[:MAX_REGEX_LINE_LEN]
1900
+ for match in _safe_regex_matches(pattern, line):
1901
+ column = match.start() + 1
1902
+ snippet = line.strip()
1903
+ if len(snippet) > 200:
1904
+ snippet = snippet[:197] + "..."
1905
+ yield Finding(
1906
+ id=rule.id,
1907
+ rule_name=rule.name,
1908
+ severity=rule.severity,
1909
+ category=rule.category,
1910
+ file=ctx.path,
1911
+ line=line_no,
1912
+ column=column,
1913
+ snippet=snippet,
1914
+ message=rule.message,
1915
+ remediation=rule.remediation,
1916
+ language=ctx.language,
1917
+ confidence=rule.confidence,
1918
+ fix_before=rule.fix_before,
1919
+ fix_after=rule.fix_after,
1920
+ action=rule.action,
1921
+ )
1922
+
1923
+
1924
+ def _display_path(path: str) -> str:
1925
+ """Return a path relative to CWD when possible, otherwise the original."""
1926
+ try:
1927
+ p = Path(path).resolve()
1928
+ cwd = Path.cwd().resolve()
1929
+ rel = p.relative_to(cwd)
1930
+ return str(rel) if str(rel) != "." else "."
1931
+ except (ValueError, OSError):
1932
+ return path
1933
+
1934
+
1935
+ def scan_file(
1936
+ path: Path,
1937
+ rules: Sequence[Rule],
1938
+ max_size: int = DEFAULT_MAX_SIZE,
1939
+ use_cache: bool = False,
1940
+ rules_fingerprint: str = "",
1941
+ ) -> list[Finding]:
1942
+ language = _detect_language(path)
1943
+ if language is None:
1944
+ return []
1945
+ display = _display_path(str(path))
1946
+ try:
1947
+ size = path.stat().st_size
1948
+ if size > max_size:
1949
+ return [_make_meta_finding(
1950
+ display, language,
1951
+ f"File skipped: size {size} bytes exceeds limit {max_size} bytes.",
1952
+ )]
1953
+ # utf-8-sig auto-strips a leading BOM if present.
1954
+ source = path.read_text(encoding="utf-8-sig", errors="replace")
1955
+ except OSError as e:
1956
+ return [_make_meta_finding(
1957
+ display, language, f"File skipped: {e}",
1958
+ )]
1959
+ if use_cache:
1960
+ cached = cache_load(source, language, rules_fingerprint)
1961
+ if cached is not None:
1962
+ # Patch path in cached findings (in case file moved)
1963
+ return [
1964
+ Finding(
1965
+ id=f.id, rule_name=f.rule_name, severity=f.severity,
1966
+ category=f.category, file=display, line=f.line,
1967
+ column=f.column, snippet=f.snippet, message=f.message,
1968
+ remediation=f.remediation, language=f.language,
1969
+ confidence=f.confidence,
1970
+ )
1971
+ for f in cached
1972
+ ]
1973
+ findings = scan_source(source, language, rules, path=display)
1974
+ if use_cache:
1975
+ cache_store(source, language, rules_fingerprint, findings)
1976
+ return findings
1977
+
1978
+
1979
+ def _detect_language(path: Path) -> Language | None:
1980
+ suffix = path.suffix.lower()
1981
+ if suffix == ".py":
1982
+ return Language.PYTHON
1983
+ if suffix in (".js", ".mjs", ".cjs"):
1984
+ return Language.JAVASCRIPT
1985
+ return None
1986
+
1987
+
1988
+ # =============================================================================
1989
+ # [SECTION] Reporters
1990
+ # =============================================================================
1991
+
1992
+ _ANSI = {
1993
+ Severity.CRITICAL: "\033[1;31m",
1994
+ Severity.HIGH: "\033[31m",
1995
+ Severity.MEDIUM: "\033[33m",
1996
+ Severity.LOW: "\033[90m",
1997
+ Severity.INFO: "\033[36m",
1998
+ }
1999
+ _ANSI_RESET = "\033[0m"
2000
+
2001
+
2002
+ def _color_enabled(force_no_color: bool) -> bool:
2003
+ if force_no_color:
2004
+ return False
2005
+ if os.environ.get("NO_COLOR"):
2006
+ return False
2007
+ try:
2008
+ return sys.stdout.isatty()
2009
+ except Exception:
2010
+ return False
2011
+
2012
+
2013
+ # --- Visual tokens (Unicode safe on modern terminals) ---
2014
+ _BOX_TL = "\u256d" # rounded corner top-left
2015
+ _BOX_TR = "\u256e"
2016
+ _BOX_BL = "\u2570"
2017
+ _BOX_BR = "\u256f"
2018
+ _BOX_H = "\u2500"
2019
+ _BOX_V = "\u2502"
2020
+ _BOX_BAR = "\u258e" # left bar accent
2021
+
2022
+ _ICON_FINDING = "\u2716" # heavy multiply (finding)
2023
+ _ICON_ARROW = "\u21b3" # arrow for remediation
2024
+ _ICON_CLOCK = "\u23f1" # stopwatch
2025
+ _ICON_DOT = "\u25cf" # filled circle
2026
+ _ICON_CHECK = "\u2714" # check mark for clean
2027
+
2028
+ _SEV_LABEL = {
2029
+ Severity.CRITICAL: "CRITICAL",
2030
+ Severity.HIGH: "HIGH",
2031
+ Severity.MEDIUM: "MEDIUM",
2032
+ Severity.LOW: "LOW",
2033
+ Severity.INFO: "INFO",
2034
+ }
2035
+
2036
+
2037
+ def _c(text: str, sev: Severity, use_color: bool) -> str:
2038
+ if not use_color or sev not in _ANSI:
2039
+ return text
2040
+ return f"{_ANSI[sev]}{text}{_ANSI_RESET}"
2041
+
2042
+
2043
+ import unicodedata as _unicodedata
2044
+
2045
+ _WIDE_RANGES = (
2046
+ (0x1100, 0x115F), (0x2E80, 0x303E), (0x3041, 0x33FF),
2047
+ (0x3400, 0x4DBF), (0x4E00, 0x9FFF), (0xA000, 0xA4CF),
2048
+ (0xAC00, 0xD7A3), (0xF900, 0xFAFF), (0xFE30, 0xFE4F),
2049
+ (0xFF00, 0xFF60), (0xFFE0, 0xFFE6),
2050
+ (0x1F300, 0x1F64F), (0x1F900, 0x1F9FF),
2051
+ )
2052
+
2053
+
2054
+ def _char_width(ch: str) -> int:
2055
+ code = ord(ch)
2056
+ for lo, hi in _WIDE_RANGES:
2057
+ if lo <= code <= hi:
2058
+ return 2
2059
+ if _unicodedata.east_asian_width(ch) in ("W", "F"):
2060
+ return 2
2061
+ return 1
2062
+
2063
+
2064
+ def _visible_width(text: str) -> int:
2065
+ import re as _re
2066
+ plain = _re.sub(r"\x1b\[[0-9;]*m", "", text)
2067
+ return sum(_char_width(c) for c in plain)
2068
+
2069
+
2070
+ def _pad_visible(text: str, width: int) -> str:
2071
+ w = _visible_width(text)
2072
+ pad = max(0, width - w)
2073
+ return text + " " * pad
2074
+
2075
+
2076
+ def _hr(width: int = 64, char: str = _BOX_H) -> str:
2077
+ return char * width
2078
+
2079
+
2080
+ def _box_top(width: int) -> str:
2081
+ return _BOX_TL + _BOX_H * (width - 2) + _BOX_TR
2082
+
2083
+
2084
+ def _box_bot(width: int) -> str:
2085
+ return _BOX_BL + _BOX_H * (width - 2) + _BOX_BR
2086
+
2087
+
2088
+ def _box_row(text: str, width: int) -> str:
2089
+ return f"{_BOX_V} {_pad_visible(text, width - 3)}{_BOX_V}"
2090
+
2091
+
2092
+ def report_cli(findings: Sequence[Finding],
2093
+ files_scanned: int,
2094
+ duration_ms: int,
2095
+ use_color: bool,
2096
+ quiet: bool = False) -> str:
2097
+ W = 64
2098
+ out: list[str] = []
2099
+
2100
+ # --- Header box ---
2101
+ brand = _c("CodeFence", Severity.INFO, use_color)
2102
+ ver = f"v{TOOL_VERSION}"
2103
+ line1 = f" {brand} {_c('\u00b7', Severity.LOW, use_color)} {ver}"
2104
+ line2 = " Pattern-based sanity check \u00b7 not a security audit"
2105
+ out.append(_box_top(W))
2106
+ out.append(_box_row(line1, W))
2107
+ out.append(_box_row(line2, W))
2108
+ out.append(_box_bot(W))
2109
+ out.append("")
2110
+
2111
+ if quiet:
2112
+ # quiet: only summary
2113
+ summary = _summary_counts(findings)
2114
+ out.append(f" Scanned {files_scanned} file(s) in {duration_ms} ms")
2115
+ out.append(_format_summary_inline(summary, use_color))
2116
+ return "\n".join(out)
2117
+
2118
+ # --- Scan info ---
2119
+ clock = _c(_ICON_CLOCK, Severity.INFO, use_color)
2120
+ plural = "file" if files_scanned == 1 else "files"
2121
+ out.append(
2122
+ f" {clock} Scanned {files_scanned} {plural} "
2123
+ f"\u00b7 {duration_ms} ms"
2124
+ )
2125
+ out.append("")
2126
+
2127
+ if not findings:
2128
+ if files_scanned == 0:
2129
+ warn = _c("No files matched.", Severity.MEDIUM, use_color)
2130
+ out.append(f" {warn}")
2131
+ out.append(
2132
+ " Check your paths, --include, and --exclude options."
2133
+ )
2134
+ else:
2135
+ check = _c(_ICON_CHECK, Severity.INFO, use_color)
2136
+ out.append(
2137
+ f" {check} {_c('No findings.', Severity.LOW, use_color)}"
2138
+ )
2139
+ out.append("")
2140
+ out.append(_summary_box(findings, W, use_color))
2141
+ return "\n".join(out)
2142
+
2143
+ # --- Group by severity, ordered ---
2144
+ order = [Severity.CRITICAL, Severity.HIGH, Severity.MEDIUM,
2145
+ Severity.LOW, Severity.INFO]
2146
+ groups: dict[Severity, list[Finding]] = {s: [] for s in order}
2147
+ for f in findings:
2148
+ groups[f.severity].append(f)
2149
+
2150
+ for sev in order:
2151
+ bucket = groups[sev]
2152
+ if not bucket:
2153
+ continue
2154
+ label = _SEV_LABEL[sev]
2155
+ count = len(bucket)
2156
+ count_word = "finding" if count == 1 else "findings"
2157
+ header = (f" {_c(_BOX_BAR, sev, use_color)}"
2158
+ f"{_c(label, sev, use_color)} "
2159
+ f"{_c(str(count), sev, use_color)} {count_word}")
2160
+ out.append(header)
2161
+ out.append(" " + _hr(W - 2))
2162
+ for f in bucket:
2163
+ mark = _c(_ICON_FINDING, sev, use_color)
2164
+ rid = _c(f.id, sev, use_color)
2165
+ out.append(f" {mark} {rid} {f.rule_name}")
2166
+ out.append(f" {_c(f.file, Severity.LOW, use_color)}:"
2167
+ f"{f.line}:{f.column}")
2168
+ out.append(f" {f.message}")
2169
+ if f.snippet:
2170
+ snip = f.snippet
2171
+ if len(snip) > 100:
2172
+ snip = snip[:97] + "..."
2173
+ out.append(f" {_c('>', Severity.LOW, use_color)} {snip}")
2174
+ if f.remediation:
2175
+ out.append(f" {_c(_ICON_ARROW, Severity.INFO, use_color)}"
2176
+ f" {_c(f.remediation, Severity.LOW, use_color)}")
2177
+ if f.fix_before or f.fix_after:
2178
+ out.append("")
2179
+ out.append(f" {_c('Typical fix:', Severity.INFO, use_color)}")
2180
+ for line in f.fix_before.split("\n"):
2181
+ prefix = _c(" - ", Severity.CRITICAL, use_color)
2182
+ out.append(f" {prefix}{line}")
2183
+ for line in f.fix_after.split("\n"):
2184
+ prefix = _c(" + ", Severity.INFO, use_color)
2185
+ out.append(f" {prefix}{line}")
2186
+ out.append("")
2187
+
2188
+ out.append(_summary_box(findings, W, use_color))
2189
+ return "\n".join(out)
2190
+
2191
+
2192
+ def _summary_box(findings: Sequence[Finding], width: int,
2193
+ use_color: bool) -> str:
2194
+ counts = _summary_counts(findings)
2195
+ out: list[str] = []
2196
+ title = _c(" Summary ", Severity.INFO, use_color)
2197
+ dash_len = width - 4 - len(" Summary ")
2198
+ top = (f" {_BOX_TL}{_BOX_H} {title}{_BOX_H * dash_len}"
2199
+ f"{_BOX_TR}")
2200
+ out.append(top)
2201
+ # two rows: critical/high and medium/low (+ info if present)
2202
+ def cell(sev: Severity, count: int) -> str:
2203
+ dot = _c(_ICON_DOT, sev, use_color)
2204
+ name = _SEV_LABEL[sev].lower()
2205
+ num = _c(f"{count:>3}", sev, use_color)
2206
+ return f"{dot} {name:<9} {num}"
2207
+ has_info = counts.get("info", 0) > 0
2208
+ row1 = f" {cell(Severity.CRITICAL, counts.get('critical', 0))}" \
2209
+ f" {cell(Severity.HIGH, counts.get('high', 0))}"
2210
+ row2 = f" {cell(Severity.MEDIUM, counts.get('medium', 0))}" \
2211
+ f" {cell(Severity.LOW, counts.get('low', 0))}"
2212
+ out.append(f" {_BOX_V} {_pad_visible(row1[2:], width - 4)}{_BOX_V}")
2213
+ out.append(f" {_BOX_V} {_pad_visible(row2[2:], width - 4)}{_BOX_V}")
2214
+ if has_info:
2215
+ row3 = f" {cell(Severity.INFO, counts.get('info', 0))}"
2216
+ out.append(f" {_BOX_V} {_pad_visible(row3[2:], width - 4)}{_BOX_V}")
2217
+ out.append(f" {_box_bot(width - 2)}")
2218
+ return "\n".join(out)
2219
+
2220
+
2221
+ def _format_summary_inline(counts: dict[str, int],
2222
+ use_color: bool) -> str:
2223
+ parts = []
2224
+ for sev in (Severity.CRITICAL, Severity.HIGH, Severity.MEDIUM,
2225
+ Severity.LOW, Severity.INFO):
2226
+ n = counts.get(sev.value, 0)
2227
+ if n == 0 and sev == Severity.INFO:
2228
+ continue
2229
+ parts.append(f"{_c(sev.value, sev, use_color)}={n}")
2230
+ return "Findings: " + ", ".join(parts)
2231
+
2232
+
2233
+ def _summary_counts(findings: Sequence[Finding]) -> dict[str, int]:
2234
+ counts: dict[str, int] = {}
2235
+ for f in findings:
2236
+ counts[f.severity.value] = counts.get(f.severity.value, 0) + 1
2237
+ return counts
2238
+
2239
+
2240
+ def report_json(findings: Sequence[Finding],
2241
+ files_scanned: int,
2242
+ duration_ms: int) -> str:
2243
+ payload = {
2244
+ "tool": TOOL_NAME,
2245
+ "version": TOOL_VERSION,
2246
+ "scan": {
2247
+ "files": files_scanned,
2248
+ "duration_ms": duration_ms,
2249
+ "timestamp": datetime.now(timezone.utc).isoformat(),
2250
+ },
2251
+ "findings": [
2252
+ {
2253
+ "id": f.id,
2254
+ "rule_name": f.rule_name,
2255
+ "severity": f.severity.value,
2256
+ "category": f.category,
2257
+ "file": f.file,
2258
+ "line": f.line,
2259
+ "column": f.column,
2260
+ "snippet": f.snippet,
2261
+ "message": f.message,
2262
+ "remediation": f.remediation,
2263
+ "language": f.language.value,
2264
+ "confidence": f.confidence.value,
2265
+ "fix_before": f.fix_before,
2266
+ "fix_after": f.fix_after,
2267
+ }
2268
+ for f in findings
2269
+ ],
2270
+ "summary": _summary_counts(findings),
2271
+ }
2272
+ return json.dumps(payload, indent=2, ensure_ascii=False)
2273
+
2274
+
2275
+ SEVERITY_TO_SARIF = {
2276
+ Severity.CRITICAL: ("error", "9.0"),
2277
+ Severity.HIGH: ("error", "7.5"),
2278
+ Severity.MEDIUM: ("warning", "5.0"),
2279
+ Severity.LOW: ("note", "3.0"),
2280
+ Severity.INFO: ("note", "0.0"),
2281
+ }
2282
+
2283
+
2284
+ def report_sarif(rules: Sequence[Rule],
2285
+ findings: Sequence[Finding]) -> str:
2286
+ used_rule_ids = {f.id for f in findings}
2287
+ sarif_rules = []
2288
+ for r in rules:
2289
+ if r.id not in used_rule_ids and r.id != "R000":
2290
+ continue
2291
+ level, score = SEVERITY_TO_SARIF.get(r.severity, ("warning", "0.0"))
2292
+ rule_props = {
2293
+ "security-severity": score,
2294
+ "tags": [r.category] if r.category else [],
2295
+ "codefence/rule_version": r.rule_version,
2296
+ }
2297
+ if r.fingerprint:
2298
+ rule_props["codefence/rule_fingerprint"] = r.fingerprint
2299
+ sarif_rules.append({
2300
+ "id": r.id,
2301
+ "name": r.name,
2302
+ "shortDescription": {"text": r.name},
2303
+ "fullDescription": {"text": r.description},
2304
+ "defaultConfiguration": {"level": level},
2305
+ "helpUri": "https://github.com/codefence",
2306
+ "properties": rule_props,
2307
+ })
2308
+ if "R000" in used_rule_ids:
2309
+ sarif_rules.append({
2310
+ "id": "R000",
2311
+ "name": "File not scanned",
2312
+ "shortDescription": {"text": "File not scanned"},
2313
+ "defaultConfiguration": {"level": "note"},
2314
+ })
2315
+
2316
+ results = []
2317
+ for f in findings:
2318
+ level, _score = SEVERITY_TO_SARIF.get(f.severity, ("warning", "0.0"))
2319
+ fp = _finding_fingerprint(f)
2320
+ results.append({
2321
+ "ruleId": f.id,
2322
+ "level": level,
2323
+ "message": {"text": f.message},
2324
+ "locations": [{
2325
+ "physicalLocation": {
2326
+ "artifactLocation": {"uri": f.file},
2327
+ "region": {
2328
+ "startLine": max(1, f.line),
2329
+ "startColumn": max(1, f.column),
2330
+ "snippet": {"text": f.snippet or ""},
2331
+ },
2332
+ },
2333
+ }],
2334
+ "partialFingerprints": {
2335
+ "codefence/v1": fp,
2336
+ },
2337
+ "properties": {
2338
+ "confidence": f.confidence.value,
2339
+ "remediation": f.remediation,
2340
+ },
2341
+ })
2342
+
2343
+ payload = {
2344
+ "$schema": "https://json.schemastore.org/sarif-2.1.0.json",
2345
+ "version": "2.1.0",
2346
+ "runs": [{
2347
+ "tool": {
2348
+ "driver": {
2349
+ "name": TOOL_NAME,
2350
+ "version": TOOL_VERSION,
2351
+ "informationUri": "https://github.com/codefence",
2352
+ "rules": sarif_rules,
2353
+ },
2354
+ },
2355
+ "results": results,
2356
+ }],
2357
+ }
2358
+ return json.dumps(payload, indent=2, ensure_ascii=False)
2359
+
2360
+
2361
+ _HTML_CSS = """
2362
+ :root{
2363
+ --bg:#f6f8fb;--card:#ffffff;--fg:#1a1f2e;--muted:#64748b;
2364
+ --border:#e2e8f0;--code-bg:#0f172a;--code-fg:#e2e8f0;
2365
+ --crit:#dc2626;--high:#ea580c;--med:#d97706;--low:#64748b;--info:#0891b2;
2366
+ --shadow:0 1px 3px rgba(15,23,42,.08),0 1px 2px rgba(15,23,42,.04);
2367
+ }
2368
+ [data-theme="dark"]{
2369
+ --bg:#0b1020;--card:#131a2e;--fg:#e8ecf5;--muted:#94a3b8;
2370
+ --border:#1f2a44;--code-bg:#050912;--code-fg:#cbd5e1;
2371
+ --crit:#f87171;--high:#fb923c;--med:#fbbf24;--low:#94a3b8;--info:#22d3ee;
2372
+ --shadow:0 1px 3px rgba(0,0,0,.4),0 1px 2px rgba(0,0,0,.3);
2373
+ }
2374
+ *{box-sizing:border-box}
2375
+ html,body{margin:0;padding:0}
2376
+ body{
2377
+ padding:32px 20px 60px;font-family:-apple-system,BlinkMacSystemFont,
2378
+ "Segoe UI",Roboto,"Helvetica Neue",Arial,sans-serif;
2379
+ background:var(--bg);color:var(--fg);line-height:1.5;
2380
+ transition:background .2s,color .2s;
2381
+ }
2382
+ .container{max-width:960px;margin:0 auto}
2383
+ header{margin-bottom:24px;display:flex;justify-content:space-between;
2384
+ align-items:flex-start;gap:16px;flex-wrap:wrap}
2385
+ h1{margin:0 0 4px;font-size:26px;font-weight:700;letter-spacing:-.02em}
2386
+ .tagline{color:var(--muted);font-size:14px;margin:0}
2387
+ .meta{color:var(--muted);font-size:13px;margin-top:8px}
2388
+ button.theme-btn{
2389
+ padding:6px 11px;border:1px solid var(--border);background:var(--card);
2390
+ color:var(--fg);border-radius:8px;cursor:pointer;font-size:12px;
2391
+ font-weight:500;box-shadow:var(--shadow);transition:transform .15s;
2392
+ white-space:nowrap;flex-shrink:0;
2393
+ }
2394
+ button.theme-btn:hover{transform:translateY(-1px)}
2395
+ .dashboard{
2396
+ display:grid;grid-template-columns:repeat(auto-fit,minmax(140px,1fr));
2397
+ gap:12px;margin-bottom:24px;
2398
+ }
2399
+ .stat{
2400
+ background:var(--card);border:1px solid var(--border);border-radius:12px;
2401
+ padding:16px 18px;box-shadow:var(--shadow);
2402
+ }
2403
+ .stat .label{font-size:11px;font-weight:600;text-transform:uppercase;
2404
+ letter-spacing:.08em;color:var(--muted);margin-bottom:4px}
2405
+ .stat .value{font-size:28px;font-weight:700;letter-spacing:-.03em;line-height:1}
2406
+ .stat.crit .value{color:var(--crit)}
2407
+ .stat.high .value{color:var(--high)}
2408
+ .stat.med .value{color:var(--med)}
2409
+ .stat.low .value{color:var(--low)}
2410
+ .stat.info .value{color:var(--info)}
2411
+ .stat.total .value{color:var(--fg)}
2412
+ h2.section{
2413
+ font-size:13px;font-weight:700;text-transform:uppercase;
2414
+ letter-spacing:.1em;color:var(--muted);margin:28px 0 12px;
2415
+ display:flex;align-items:center;gap:10px;
2416
+ }
2417
+ h2.section::after{
2418
+ content:"";flex:1;height:1px;background:var(--border);
2419
+ }
2420
+ .card{
2421
+ background:var(--card);border:1px solid var(--border);border-radius:12px;
2422
+ padding:16px 18px;margin-bottom:10px;box-shadow:var(--shadow);
2423
+ border-left:4px solid var(--border);
2424
+ }
2425
+ .card.critical{border-left-color:var(--crit)}
2426
+ .card.high{border-left-color:var(--high)}
2427
+ .card.medium{border-left-color:var(--med)}
2428
+ .card.low{border-left-color:var(--low)}
2429
+ .card.info{border-left-color:var(--info)}
2430
+ .card-head{display:flex;align-items:center;gap:10px;margin-bottom:8px;
2431
+ flex-wrap:wrap}
2432
+ .badge{
2433
+ display:inline-block;padding:3px 9px;border-radius:999px;
2434
+ font-size:11px;font-weight:700;letter-spacing:.04em;text-transform:uppercase;
2435
+ }
2436
+ .badge.critical{background:var(--crit);color:#fff}
2437
+ .badge.high{background:var(--high);color:#fff}
2438
+ .badge.medium{background:var(--med);color:#fff}
2439
+ .badge.low{background:var(--low);color:#fff}
2440
+ .badge.info{background:var(--info);color:#fff}
2441
+ .rule-id{font-family:ui-monospace,"SF Mono",Menlo,Consolas,monospace;
2442
+ font-size:12px;font-weight:700;color:var(--muted);
2443
+ padding:3px 7px;background:var(--bg);border-radius:6px}
2444
+ .rule-name{font-weight:600;font-size:15px}
2445
+ .location{
2446
+ font-family:ui-monospace,"SF Mono",Menlo,Consolas,monospace;
2447
+ font-size:12px;color:var(--muted);margin-bottom:8px;
2448
+ word-break:break-all;
2449
+ }
2450
+ .message{font-size:14px;margin-bottom:10px}
2451
+ pre.snippet{
2452
+ background:var(--code-bg);color:var(--code-fg);
2453
+ padding:10px 12px;border-radius:8px;overflow-x:auto;
2454
+ font-family:ui-monospace,"SF Mono",Menlo,Consolas,monospace;
2455
+ font-size:12.5px;line-height:1.5;margin:0 0 8px;white-space:pre-wrap;
2456
+ word-break:break-word;
2457
+ }
2458
+ .remediation{
2459
+ font-size:13px;color:var(--muted);
2460
+ padding:8px 12px;background:var(--bg);border-radius:8px;
2461
+ border-left:3px solid var(--info);
2462
+ margin-bottom:10px;
2463
+ }
2464
+ .remediation strong{color:var(--fg);font-weight:600}
2465
+ .fix-example{
2466
+ margin-top:10px;
2467
+ background:var(--bg);
2468
+ border:1px solid var(--border);
2469
+ border-radius:8px;
2470
+ padding:10px 12px;
2471
+ }
2472
+ .fix-title{
2473
+ font-size:11px;font-weight:700;text-transform:uppercase;
2474
+ letter-spacing:.08em;color:var(--muted);margin-bottom:8px;
2475
+ }
2476
+ pre.fix-before,pre.fix-after{
2477
+ margin:0 0 6px;
2478
+ padding:8px 10px;
2479
+ border-radius:6px;
2480
+ font-family:ui-monospace,"SF Mono",Menlo,Consolas,monospace;
2481
+ font-size:12.5px;line-height:1.5;
2482
+ white-space:pre-wrap;word-break:break-word;
2483
+ }
2484
+ pre.fix-before{
2485
+ background:rgba(220,38,38,.08);
2486
+ color:var(--fg);
2487
+ border-left:3px solid var(--crit);
2488
+ }
2489
+ pre.fix-after{
2490
+ background:rgba(8,145,178,.10);
2491
+ color:var(--fg);
2492
+ border-left:3px solid var(--info);
2493
+ margin-bottom:0;
2494
+ }
2495
+ pre.fix-before .ln,pre.fix-after .ln{
2496
+ color:var(--muted);
2497
+ font-weight:700;
2498
+ user-select:none;
2499
+ }
2500
+ .fix-after-wrap{
2501
+ position:relative;
2502
+ padding-top:30px;
2503
+ }
2504
+ .copy-btn{
2505
+ position:absolute;
2506
+ top:0;right:0;
2507
+ display:inline-flex;align-items:center;gap:5px;
2508
+ padding:4px 10px;
2509
+ font-size:11px;font-weight:600;
2510
+ border:1px solid var(--border);
2511
+ background:var(--card);
2512
+ color:var(--fg);
2513
+ border-radius:6px;
2514
+ cursor:pointer;
2515
+ opacity:.95;
2516
+ transition:opacity .15s,transform .15s,background .15s;
2517
+ }
2518
+ .copy-btn:hover{opacity:1;transform:translateY(-1px)}
2519
+ .copy-btn:active{transform:translateY(0)}
2520
+ .copy-btn.copied{
2521
+ background:var(--info);color:#fff;border-color:var(--info);
2522
+ }
2523
+ .copy-btn .copy-icon{font-size:13px;line-height:1}
2524
+ @media (max-width:600px){
2525
+ .fix-after-wrap{padding-top:28px}
2526
+ .copy-btn{padding:5px 9px;font-size:11px}
2527
+ }
2528
+ .empty{
2529
+ text-align:center;padding:60px 20px;color:var(--muted);
2530
+ background:var(--card);border:1px dashed var(--border);border-radius:12px;
2531
+ }
2532
+ .empty .icon{font-size:48px;margin-bottom:12px;display:block}
2533
+ footer{
2534
+ margin-top:40px;text-align:center;color:var(--muted);font-size:12px;
2535
+ padding-top:20px;border-top:1px solid var(--border);
2536
+ }
2537
+ @media (max-width:600px){
2538
+ body{padding:20px 14px 40px}
2539
+ h1{font-size:22px}
2540
+ .stat .value{font-size:22px}
2541
+ }
2542
+ """
2543
+
2544
+ _HTML_JS = r"""
2545
+ (function(){
2546
+ var root = document.documentElement;
2547
+ var btn = document.getElementById('theme');
2548
+ var saved = null;
2549
+ try { saved = localStorage.getItem('ai-sanitizer-theme'); } catch(e) {}
2550
+ if (!saved) {
2551
+ saved = (window.matchMedia &&
2552
+ window.matchMedia('(prefers-color-scheme: dark)').matches)
2553
+ ? 'dark' : 'light';
2554
+ }
2555
+ root.setAttribute('data-theme', saved);
2556
+
2557
+ function updateLabel() {
2558
+ btn.textContent = root.getAttribute('data-theme') === 'dark'
2559
+ ? 'Light mode' : 'Dark mode';
2560
+ }
2561
+ updateLabel();
2562
+
2563
+ btn.addEventListener('click', function() {
2564
+ var cur = root.getAttribute('data-theme') === 'dark' ? 'light' : 'dark';
2565
+ root.setAttribute('data-theme', cur);
2566
+ try { localStorage.setItem('ai-sanitizer-theme', cur); } catch(e) {}
2567
+ updateLabel();
2568
+ });
2569
+
2570
+ function getPlainFix(el) {
2571
+ var text = el.innerText || el.textContent || '';
2572
+ return text.replace(/^\+ ?/gm, '').trim();
2573
+ }
2574
+
2575
+ function fallbackCopy(text) {
2576
+ var ta = document.createElement('textarea');
2577
+ ta.value = text;
2578
+ ta.style.position = 'fixed';
2579
+ ta.style.top = '-1000px';
2580
+ ta.style.opacity = '0';
2581
+ document.body.appendChild(ta);
2582
+ ta.select();
2583
+ try { document.execCommand('copy'); } catch(e) {}
2584
+ document.body.removeChild(ta);
2585
+ }
2586
+
2587
+ function markCopied(btn) {
2588
+ var t = btn.querySelector('.copy-text');
2589
+ var old = t ? t.textContent : '';
2590
+ btn.classList.add('copied');
2591
+ if (t) { t.textContent = 'Copied'; }
2592
+ setTimeout(function() {
2593
+ btn.classList.remove('copied');
2594
+ if (t) { t.textContent = old || 'Copy'; }
2595
+ }, 1400);
2596
+ }
2597
+
2598
+ document.querySelectorAll('.copy-btn').forEach(function(btn) {
2599
+ btn.addEventListener('click', function() {
2600
+ var id = btn.getAttribute('data-target');
2601
+ var el = document.getElementById(id);
2602
+ if (!el) return;
2603
+ var text = getPlainFix(el);
2604
+ if (navigator.clipboard && navigator.clipboard.writeText) {
2605
+ navigator.clipboard.writeText(text).then(function() {
2606
+ markCopied(btn);
2607
+ }).catch(function() {
2608
+ fallbackCopy(text); markCopied(btn);
2609
+ });
2610
+ } else {
2611
+ fallbackCopy(text); markCopied(btn);
2612
+ }
2613
+ });
2614
+ });
2615
+ })();
2616
+ """
2617
+
2618
+
2619
+ def report_html(findings: Sequence[Finding],
2620
+ files_scanned: int,
2621
+ duration_ms: int) -> str:
2622
+ esc = _html.escape
2623
+ counts = _summary_counts(findings)
2624
+ total = len([f for f in findings if f.severity != Severity.INFO])
2625
+
2626
+ def stat(cls: str, label: str, value: int) -> str:
2627
+ return (
2628
+ f'<div class="stat {cls}">'
2629
+ f'<div class="label">{esc(label)}</div>'
2630
+ f'<div class="value">{value}</div>'
2631
+ f'</div>'
2632
+ )
2633
+
2634
+ dashboard = (
2635
+ stat("total", "Total", total)
2636
+ + stat("crit", "Critical", counts.get("critical", 0))
2637
+ + stat("high", "High", counts.get("high", 0))
2638
+ + stat("med", "Medium", counts.get("medium", 0))
2639
+ + stat("low", "Low", counts.get("low", 0))
2640
+ + (stat("info", "Info", counts.get("info", 0))
2641
+ if counts.get("info", 0) else "")
2642
+ )
2643
+
2644
+ if not findings:
2645
+ body = (
2646
+ '<div class="empty">'
2647
+ '<span class="icon">\u2714</span>'
2648
+ '<strong>No findings.</strong><br>'
2649
+ 'The scanned file(s) look clean according to the active rules.'
2650
+ '</div>'
2651
+ )
2652
+ else:
2653
+ # group by severity
2654
+ order = [Severity.CRITICAL, Severity.HIGH, Severity.MEDIUM,
2655
+ Severity.LOW, Severity.INFO]
2656
+ labels = {
2657
+ Severity.CRITICAL: "Critical",
2658
+ Severity.HIGH: "High",
2659
+ Severity.MEDIUM: "Medium",
2660
+ Severity.LOW: "Low",
2661
+ Severity.INFO: "Info",
2662
+ }
2663
+ sections: list[str] = []
2664
+ for sev in order:
2665
+ bucket = [f for f in findings if f.severity == sev]
2666
+ if not bucket:
2667
+ continue
2668
+ plural = "finding" if len(bucket) == 1 else "findings"
2669
+ sections.append(
2670
+ f'<h2 class="section">{esc(labels[sev])} '
2671
+ f'&middot; {len(bucket)} {plural}</h2>'
2672
+ )
2673
+ for f in bucket:
2674
+ snippet = (
2675
+ f'<pre class="snippet">{esc(f.snippet)}</pre>'
2676
+ if f.snippet else ""
2677
+ )
2678
+ rem = (
2679
+ f'<div class="remediation">'
2680
+ f'<strong>Fix:</strong> {esc(f.remediation)}</div>'
2681
+ if f.remediation else ""
2682
+ )
2683
+ fix_block = ""
2684
+ if f.fix_before or f.fix_after:
2685
+ # Unique id for the copy target
2686
+ safe_id = f"fix-{f.id}-{f.line}-{f.column}".replace(
2687
+ ".", "_")
2688
+ parts = ['<div class="fix-example">',
2689
+ '<div class="fix-title">Typical fix</div>']
2690
+ if f.fix_before:
2691
+ parts.append('<pre class="fix-before">')
2692
+ for ln in f.fix_before.split("\n"):
2693
+ parts.append(f'<span class="ln">- </span>'
2694
+ f'{esc(ln)}')
2695
+ parts.append('</pre>')
2696
+ if f.fix_after:
2697
+ parts.append(
2698
+ '<div class="fix-after-wrap">'
2699
+ '<button type="button" class="copy-btn" '
2700
+ f'data-target="{safe_id}" '
2701
+ 'aria-label="Copy fix">'
2702
+ '<span class="copy-icon">&#x2398;</span>'
2703
+ '<span class="copy-text">Copy</span>'
2704
+ '</button>'
2705
+ f'<pre class="fix-after" id="{safe_id}">'
2706
+ )
2707
+ for ln in f.fix_after.split("\n"):
2708
+ parts.append(f'<span class="ln">+ </span>'
2709
+ f'{esc(ln)}')
2710
+ parts.append('</pre>')
2711
+ parts.append('</div>')
2712
+ parts.append('</div>')
2713
+ fix_block = "".join(parts)
2714
+ sections.append(
2715
+ f'<div class="card {f.severity.value}">'
2716
+ f'<div class="card-head">'
2717
+ f'<span class="badge {f.severity.value}">'
2718
+ f'{esc(f.severity.value)}</span>'
2719
+ f'<span class="rule-id">{esc(f.id)}</span>'
2720
+ f'<span class="rule-name">{esc(f.rule_name)}</span>'
2721
+ f'</div>'
2722
+ f'<div class="location">{esc(f.file)}:{f.line}:{f.column}'
2723
+ f'</div>'
2724
+ f'<div class="message">{esc(f.message)}</div>'
2725
+ f'{snippet}'
2726
+ f'{rem}'
2727
+ f'{fix_block}'
2728
+ f'</div>'
2729
+ )
2730
+ body = "\n".join(sections)
2731
+
2732
+ timestamp = datetime.now(timezone.utc).strftime("%Y-%m-%d %H:%M UTC")
2733
+ plural = "file" if files_scanned == 1 else "files"
2734
+ return (
2735
+ "<!DOCTYPE html>\n"
2736
+ '<html lang="en">\n'
2737
+ "<head>\n"
2738
+ '<meta charset="utf-8">\n'
2739
+ '<meta name="viewport" content="width=device-width,initial-scale=1">\n'
2740
+ "<title>CodeFence Report</title>\n"
2741
+ f"<style>{_HTML_CSS}</style>\n"
2742
+ "</head>\n"
2743
+ "<body>\n"
2744
+ '<div class="container">\n'
2745
+ "<header>\n"
2746
+ "<div>\n"
2747
+ "<h1>CodeFence Report</h1>\n"
2748
+ '<p class="tagline">Pattern-based sanity check '
2749
+ "&middot; not a security audit</p>\n"
2750
+ f'<p class="meta">Scanned {files_scanned} {plural} '
2751
+ f'in {duration_ms} ms &middot; {esc(timestamp)}</p>\n'
2752
+ "</div>\n"
2753
+ '<button id="theme" class="theme-btn">Toggle theme</button>\n'
2754
+ "</header>\n"
2755
+ f'<div class="dashboard">{dashboard}</div>\n'
2756
+ f"{body}\n"
2757
+ "<footer>\n"
2758
+ f"Generated by {esc(TOOL_NAME)} v{esc(TOOL_VERSION)} "
2759
+ "&middot; Offline &middot; Zero network calls\n"
2760
+ "</footer>\n"
2761
+ "</div>\n"
2762
+ f"<script>{_HTML_JS}</script>\n"
2763
+ "</body>\n</html>\n"
2764
+ )
2765
+
2766
+
2767
+ # =============================================================================
2768
+ # [SECTION] CLI
2769
+ # =============================================================================
2770
+
2771
+ BASELINE_DIR = ".codefence"
2772
+ BASELINE_FILE = "baseline.json"
2773
+ BASELINE_VERSION = "v1"
2774
+ CONFIG_FILE = "config.json"
2775
+
2776
+ DEFAULT_CONFIG_CONTENT = {
2777
+ "schema": "codefence/config-v1",
2778
+ "format": "cli",
2779
+ "severity": "low",
2780
+ "include": ["*.py", "*.js", "*.mjs", "*.cjs"],
2781
+ "exclude": ["node_modules", ".git", "venv", ".venv", "__pycache__",
2782
+ "dist", "build", ".tox"],
2783
+ "max_size": 2097152,
2784
+ "cache": False,
2785
+ }
2786
+
2787
+
2788
+ def _finding_fingerprint(f: Finding) -> str:
2789
+ """Stable identity for a finding.
2790
+
2791
+ v3: includes line number and column. Two findings with the same
2792
+ rule but on different lines are now distinct, which prevents
2793
+ baseline collision when a file contains the same dangerous
2794
+ pattern multiple times.
2795
+
2796
+ Trade-off: if code shifts up or down within a file, existing
2797
+ findings may appear as "new" in --diff. This is intentional —
2798
+ catching new instances of a dangerous pattern matters more than
2799
+ avoiding re-reporting a shifted one. Use `# noqa: RXXX` for
2800
+ one-off suppressions.
2801
+ """
2802
+ h = hashlib.sha256()
2803
+ h.update(b"codefence-fp-v3\x00")
2804
+ h.update(f.id.encode("utf-8"))
2805
+ h.update(b"\x00")
2806
+ h.update(f.file.encode("utf-8"))
2807
+ h.update(b"\x00")
2808
+ h.update(str(f.line).encode("ascii"))
2809
+ h.update(b"\x00")
2810
+ h.update(str(f.column).encode("ascii"))
2811
+ h.update(b"\x00")
2812
+ snippet = (f.snippet or "").strip().lower()
2813
+ h.update(snippet.encode("utf-8"))
2814
+ return h.hexdigest()
2815
+
2816
+
2817
+ def _baseline_path(git_root: Path | None = None) -> Path:
2818
+ root = git_root or _find_git_root() or Path.cwd()
2819
+ return root / BASELINE_DIR / BASELINE_FILE
2820
+
2821
+
2822
+ def _load_baseline(path: Path) -> set[str] | None:
2823
+ """Return set of fingerprints, or None if missing/invalid."""
2824
+ try:
2825
+ if not path.is_file():
2826
+ return None
2827
+ data = json.loads(path.read_text(encoding="utf-8"))
2828
+ if not isinstance(data, dict):
2829
+ return None
2830
+ if data.get("version") != BASELINE_VERSION:
2831
+ return None
2832
+ items = data.get("findings", [])
2833
+ if not isinstance(items, list):
2834
+ return None
2835
+ out: set[str] = set()
2836
+ for item in items:
2837
+ fp = item.get("fingerprint") if isinstance(item, dict) else None
2838
+ if isinstance(fp, str):
2839
+ out.add(fp)
2840
+ return out
2841
+ except (OSError, ValueError, json.JSONDecodeError):
2842
+ return None
2843
+
2844
+
2845
+ def _save_baseline(path: Path, findings: Sequence[Finding]) -> None:
2846
+ path.parent.mkdir(parents=True, exist_ok=True)
2847
+ items = []
2848
+ seen: set[str] = set()
2849
+ for f in findings:
2850
+ fp = _finding_fingerprint(f)
2851
+ if fp in seen:
2852
+ continue
2853
+ seen.add(fp)
2854
+ items.append({
2855
+ "fingerprint": fp,
2856
+ "rule": f.id,
2857
+ "file": f.file,
2858
+ "line": f.line,
2859
+ })
2860
+ payload = {
2861
+ "version": BASELINE_VERSION,
2862
+ "created_at": datetime.now(timezone.utc).isoformat(),
2863
+ "tool_version": TOOL_VERSION,
2864
+ "findings": items,
2865
+ }
2866
+ text = json.dumps(payload, indent=2, ensure_ascii=False)
2867
+ fd, tmp_path = tempfile.mkstemp(prefix=".tmp-", dir=str(path.parent))
2868
+ try:
2869
+ with os.fdopen(fd, "w", encoding="utf-8") as fp:
2870
+ fp.write(text)
2871
+ os.replace(tmp_path, path)
2872
+ except Exception:
2873
+ try:
2874
+ os.unlink(tmp_path)
2875
+ except OSError:
2876
+ pass
2877
+ raise
2878
+
2879
+
2880
+ def _filter_new_findings(findings: Sequence[Finding],
2881
+ baseline: set[str]) -> list[Finding]:
2882
+ return [f for f in findings if _finding_fingerprint(f) not in baseline]
2883
+
2884
+
2885
+ HISTORY_DIR = "codefence"
2886
+ HISTORY_FILE = "history.db"
2887
+ HISTORY_VERSION = 1
2888
+
2889
+
2890
+ def _history_path() -> Path:
2891
+ base = os.environ.get("XDG_DATA_HOME")
2892
+ if base:
2893
+ root = Path(base) / HISTORY_DIR
2894
+ else:
2895
+ root = Path.home() / ".local" / "share" / HISTORY_DIR
2896
+ return root / HISTORY_FILE
2897
+
2898
+
2899
+ def _history_disabled() -> bool:
2900
+ val = os.environ.get("CODEFENCE_NO_HISTORY", "").strip().lower()
2901
+ return val in ("1", "true", "yes", "on")
2902
+
2903
+
2904
+ def _open_history_db() -> sqlite3.Connection | None:
2905
+ try:
2906
+ path = _history_path()
2907
+ path.parent.mkdir(parents=True, exist_ok=True, mode=0o700)
2908
+ try:
2909
+ os.chmod(path.parent, 0o700)
2910
+ except OSError:
2911
+ pass
2912
+ conn = sqlite3.connect(str(path))
2913
+ conn.execute("PRAGMA journal_mode=WAL")
2914
+ conn.execute("PRAGMA foreign_keys=ON")
2915
+ _ensure_schema(conn)
2916
+ return conn
2917
+ except (sqlite3.Error, OSError):
2918
+ return None
2919
+
2920
+
2921
+ def _ensure_schema(conn: sqlite3.Connection) -> None:
2922
+ conn.execute("""
2923
+ CREATE TABLE IF NOT EXISTS meta (
2924
+ key TEXT PRIMARY KEY,
2925
+ value TEXT NOT NULL
2926
+ )
2927
+ """)
2928
+ conn.execute("""
2929
+ CREATE TABLE IF NOT EXISTS scans (
2930
+ scan_id TEXT PRIMARY KEY,
2931
+ timestamp TEXT NOT NULL,
2932
+ tool_version TEXT NOT NULL,
2933
+ files_scanned INTEGER NOT NULL,
2934
+ duration_ms INTEGER NOT NULL,
2935
+ findings_count INTEGER NOT NULL,
2936
+ critical_count INTEGER NOT NULL,
2937
+ high_count INTEGER NOT NULL,
2938
+ medium_count INTEGER NOT NULL,
2939
+ low_count INTEGER NOT NULL,
2940
+ rules_hash TEXT NOT NULL,
2941
+ policy_hash TEXT NOT NULL,
2942
+ baseline_hash TEXT NOT NULL,
2943
+ commit_sha TEXT NOT NULL,
2944
+ result_hash TEXT NOT NULL
2945
+ )
2946
+ """)
2947
+ conn.execute("""
2948
+ CREATE TABLE IF NOT EXISTS findings (
2949
+ id INTEGER PRIMARY KEY AUTOINCREMENT,
2950
+ scan_id TEXT NOT NULL,
2951
+ rule_id TEXT NOT NULL,
2952
+ file TEXT NOT NULL,
2953
+ line INTEGER NOT NULL,
2954
+ column_no INTEGER NOT NULL,
2955
+ severity TEXT NOT NULL,
2956
+ action TEXT NOT NULL,
2957
+ message TEXT NOT NULL,
2958
+ FOREIGN KEY(scan_id) REFERENCES scans(scan_id) ON DELETE CASCADE
2959
+ )
2960
+ """)
2961
+ conn.execute(
2962
+ "CREATE INDEX IF NOT EXISTS idx_findings_rule "
2963
+ "ON findings(rule_id)"
2964
+ )
2965
+ conn.execute(
2966
+ "CREATE INDEX IF NOT EXISTS idx_scans_ts "
2967
+ "ON scans(timestamp)"
2968
+ )
2969
+ conn.execute(
2970
+ "INSERT OR REPLACE INTO meta(key, value) VALUES (?, ?)",
2971
+ ("schema_version", str(HISTORY_VERSION)),
2972
+ )
2973
+ conn.commit()
2974
+
2975
+
2976
+ def _record_scan(evidence: dict,
2977
+ findings: Sequence[Finding]) -> None:
2978
+ if _history_disabled():
2979
+ return
2980
+ conn = _open_history_db()
2981
+ if conn is None:
2982
+ return
2983
+ try:
2984
+ sev = evidence.get("by_severity", {}) or {}
2985
+ conn.execute(
2986
+ """INSERT OR REPLACE INTO scans(
2987
+ scan_id, timestamp, tool_version, files_scanned,
2988
+ duration_ms, findings_count,
2989
+ critical_count, high_count, medium_count, low_count,
2990
+ rules_hash, policy_hash, baseline_hash,
2991
+ commit_sha, result_hash
2992
+ ) VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?)""",
2993
+ (
2994
+ evidence["scan_id"],
2995
+ evidence["timestamp"],
2996
+ evidence["tool_version"],
2997
+ evidence["files_scanned"],
2998
+ evidence["duration_ms"],
2999
+ evidence["findings_count"],
3000
+ int(sev.get("critical", 0)),
3001
+ int(sev.get("high", 0)),
3002
+ int(sev.get("medium", 0)),
3003
+ int(sev.get("low", 0)),
3004
+ evidence["rules_hash"],
3005
+ evidence["policy_hash"],
3006
+ evidence["baseline_hash"],
3007
+ evidence["commit_sha"],
3008
+ evidence["result_hash"],
3009
+ ),
3010
+ )
3011
+ for f in findings:
3012
+ conn.execute(
3013
+ """INSERT INTO findings(
3014
+ scan_id, rule_id, file, line, column_no,
3015
+ severity, action, message
3016
+ ) VALUES (?,?,?,?,?,?,?,?)""",
3017
+ (
3018
+ evidence["scan_id"],
3019
+ f.id,
3020
+ f.file,
3021
+ f.line,
3022
+ f.column,
3023
+ f.severity.value,
3024
+ getattr(f, "action", "block"),
3025
+ f.message,
3026
+ ),
3027
+ )
3028
+ conn.commit()
3029
+ except (sqlite3.Error, OSError, KeyError, TypeError):
3030
+ pass
3031
+ finally:
3032
+ try:
3033
+ conn.close()
3034
+ except sqlite3.Error:
3035
+ pass
3036
+
3037
+
3038
+ def _hash_file(path: Path) -> str:
3039
+ try:
3040
+ data = path.read_bytes()
3041
+ except OSError:
3042
+ return ""
3043
+ return hashlib.sha256(data).hexdigest()
3044
+
3045
+
3046
+ def _hash_text(text: str) -> str:
3047
+ return hashlib.sha256(text.encode("utf-8")).hexdigest()
3048
+
3049
+
3050
+ def _compute_result_hash(findings: Sequence[Finding]) -> str:
3051
+ """Deterministic hash of the finding set, order-independent."""
3052
+ items = []
3053
+ for f in findings:
3054
+ items.append({
3055
+ "id": f.id,
3056
+ "file": f.file,
3057
+ "line": f.line,
3058
+ "column": f.column,
3059
+ "severity": f.severity.value,
3060
+ "action": getattr(f, "action", "block"),
3061
+ "snippet": f.snippet,
3062
+ "message": f.message,
3063
+ })
3064
+ items.sort(key=lambda x: (
3065
+ x["id"], x["file"], x["line"], x["column"], x["message"],
3066
+ ))
3067
+ canonical = json.dumps(items, sort_keys=True, ensure_ascii=False)
3068
+ return hashlib.sha256(canonical.encode("utf-8")).hexdigest()
3069
+
3070
+
3071
+ def _git_commit_sha() -> str:
3072
+ try:
3073
+ result = subprocess.run(
3074
+ ["git", "rev-parse", "HEAD"],
3075
+ capture_output=True, text=True, check=False,
3076
+ )
3077
+ except (FileNotFoundError, OSError):
3078
+ return ""
3079
+ if result.returncode != 0:
3080
+ return ""
3081
+ return result.stdout.strip()
3082
+
3083
+
3084
+ def _build_evidence(findings: Sequence[Finding],
3085
+ files_scanned: int,
3086
+ duration_ms: int,
3087
+ rules_path: Path,
3088
+ policy_path: Path | None,
3089
+ baseline_path: Path | None) -> dict:
3090
+ rules_hash = _hash_file(rules_path)
3091
+ policy_hash = _hash_file(policy_path) if policy_path else ""
3092
+ baseline_hash = _hash_file(baseline_path) if baseline_path else ""
3093
+ result_hash = _compute_result_hash(findings)
3094
+ commit_sha = _git_commit_sha()
3095
+ counts: dict[str, int] = {}
3096
+ for f in findings:
3097
+ counts[f.severity.value] = counts.get(f.severity.value, 0) + 1
3098
+ return {
3099
+ "tool": TOOL_NAME,
3100
+ "tool_version": TOOL_VERSION,
3101
+ "scan_id": result_hash[:16],
3102
+ "timestamp": datetime.now(timezone.utc).isoformat(),
3103
+ "files_scanned": files_scanned,
3104
+ "duration_ms": duration_ms,
3105
+ "rules_hash": rules_hash,
3106
+ "policy_hash": policy_hash,
3107
+ "baseline_hash": baseline_hash,
3108
+ "commit_sha": commit_sha,
3109
+ "result_hash": result_hash,
3110
+ "findings_count": len(findings),
3111
+ "by_severity": counts,
3112
+ }
3113
+
3114
+
3115
+ POLICY_SCHEMA = "codefence/policy-v1"
3116
+
3117
+
3118
+ def _load_policy(path: Path) -> dict:
3119
+ """Load and validate a policy file. Raises ValueError on error."""
3120
+ if not path.is_file():
3121
+ raise ValueError(f"policy file not found: {path}")
3122
+ data = json.loads(path.read_text(encoding="utf-8"))
3123
+ if not isinstance(data, dict):
3124
+ raise ValueError("policy file must contain a JSON object")
3125
+ schema = data.get("schema")
3126
+ if schema != POLICY_SCHEMA:
3127
+ raise ValueError(
3128
+ f"unsupported policy schema: {schema!r} (expected {POLICY_SCHEMA!r})"
3129
+ )
3130
+ overrides = data.get("overrides", {})
3131
+ if not isinstance(overrides, dict):
3132
+ raise ValueError("policy 'overrides' must be an object")
3133
+ for rid, act in overrides.items():
3134
+ if not isinstance(rid, str):
3135
+ raise ValueError("override keys must be rule IDs (strings)")
3136
+ _validate_action(act, rid)
3137
+ extra = data.get("rules", [])
3138
+ if not isinstance(extra, list):
3139
+ raise ValueError("policy 'rules' must be a list")
3140
+ for i, entry in enumerate(extra):
3141
+ if not isinstance(entry, dict):
3142
+ raise ValueError(f"policy rule {i}: must be an object")
3143
+ rid = entry.get("id", "")
3144
+ if not isinstance(rid, str) or not rid.startswith("ORG-"):
3145
+ raise ValueError(
3146
+ f"policy rule {i}: id must start with 'ORG-', got {rid!r}"
3147
+ )
3148
+ det = entry.get("detection")
3149
+ if det not in ("regex", "ast", "lexical"):
3150
+ raise ValueError(
3151
+ f"policy rule {rid}: detection must be regex/ast/lexical"
3152
+ )
3153
+ if det == "regex":
3154
+ pats = entry.get("patterns", [])
3155
+ if not isinstance(pats, list) or not pats:
3156
+ raise ValueError(
3157
+ f"policy rule {rid}: regex rules need a 'patterns' list"
3158
+ )
3159
+ for pat in pats:
3160
+ if not isinstance(pat, str):
3161
+ raise ValueError(
3162
+ f"policy rule {rid}: patterns must be strings"
3163
+ )
3164
+ if len(pat) > MAX_REGEX_PATTERN_LEN:
3165
+ raise ValueError(
3166
+ f"policy rule {rid}: pattern too long "
3167
+ f"({len(pat)} > {MAX_REGEX_PATTERN_LEN})"
3168
+ )
3169
+ try:
3170
+ re.compile(pat)
3171
+ except re.error as e:
3172
+ raise ValueError(
3173
+ f"policy rule {rid}: invalid regex: {e}"
3174
+ )
3175
+ return data
3176
+
3177
+
3178
+ def _apply_policy(rules: list[Rule], policy: dict) -> list[Rule]:
3179
+ """Return a new list of rules with policy overrides + extra rules."""
3180
+ overrides = policy.get("overrides", {}) or {}
3181
+ extra_entries = policy.get("rules", []) or []
3182
+
3183
+ # Apply overrides to existing rules (action only)
3184
+ new_rules: list[Rule] = []
3185
+ for r in rules:
3186
+ new_action = overrides.get(r.id)
3187
+ if new_action and new_action != r.action:
3188
+ new_rules.append(Rule(
3189
+ id=r.id, name=r.name, languages=r.languages,
3190
+ severity=r.severity, category=r.category,
3191
+ description=r.description, detection=r.detection,
3192
+ message=r.message, remediation=r.remediation,
3193
+ references=r.references, confidence=r.confidence,
3194
+ enabled=r.enabled, patterns=r.patterns, handler=r.handler,
3195
+ fix_before=r.fix_before, fix_after=r.fix_after,
3196
+ action=new_action,
3197
+ rule_version=r.rule_version,
3198
+ fingerprint=r.fingerprint,
3199
+ ))
3200
+ else:
3201
+ new_rules.append(r)
3202
+
3203
+ # Append extra rules
3204
+ for entry in extra_entries:
3205
+ new_rules.append(_build_rule(entry))
3206
+
3207
+ return new_rules
3208
+
3209
+
3210
+ _VALUE_FLAGS = {
3211
+ "--rules", "--policy", "--format", "--output", "--severity",
3212
+ "--max-size", "--include", "--exclude", "--config", "--evidence",
3213
+ }
3214
+
3215
+
3216
+ def _find_subcommand(argv: list[str]) -> tuple[int, str] | tuple[None, None]:
3217
+ """Find the first non-flag arg that matches a subcommand name,
3218
+ allowing flags (with values) to appear before it.
3219
+ Stops scanning as soon as a non-subcommand positional is seen."""
3220
+ i = 0
3221
+ while i < len(argv):
3222
+ a = argv[i]
3223
+ if a in _VALUE_FLAGS:
3224
+ i += 2
3225
+ continue
3226
+ if a.startswith("--") and "=" in a:
3227
+ i += 1
3228
+ continue
3229
+ if a.startswith("-") and a != "-":
3230
+ i += 1
3231
+ continue
3232
+ # First non-flag positional
3233
+ if a in SUBCOMMANDS:
3234
+ return i, a
3235
+ return None, None
3236
+ return None, None
3237
+
3238
+
3239
+ def _cmd_explain(argv: list[str]) -> int:
3240
+ """Print a structured explanation of a single rule."""
3241
+ if not argv:
3242
+ print("usage: codefence explain RXXX [--rules FILE]",
3243
+ file=sys.stderr)
3244
+ return EXIT_USAGE
3245
+
3246
+ # Parse minimal: first positional = rule id, optional --rules FILE
3247
+ rid = None
3248
+ rules_path_arg = None
3249
+ i = 0
3250
+ while i < len(argv):
3251
+ a = argv[i]
3252
+ if a == "--rules" and i + 1 < len(argv):
3253
+ rules_path_arg = argv[i + 1]
3254
+ i += 2
3255
+ continue
3256
+ if a.startswith("--rules="):
3257
+ rules_path_arg = a.split("=", 1)[1]
3258
+ i += 1
3259
+ continue
3260
+ if rid is None and not a.startswith("-"):
3261
+ rid = a
3262
+ i += 1
3263
+
3264
+ if not rid:
3265
+ print("usage: codefence explain RXXX [--rules FILE]",
3266
+ file=sys.stderr)
3267
+ return EXIT_USAGE
3268
+
3269
+ rules_path = Path(rules_path_arg) if rules_path_arg else _default_rules_path()
3270
+ if not rules_path.is_file():
3271
+ print(f"error: rules file not found: {rules_path}", file=sys.stderr)
3272
+ return EXIT_USAGE
3273
+ try:
3274
+ rules = load_rules(rules_path)
3275
+ except (OSError, ValueError, json.JSONDecodeError) as e:
3276
+ print(f"error: failed to load rules: {e}", file=sys.stderr)
3277
+ return EXIT_INTERNAL
3278
+
3279
+ rule = next((r for r in rules if r.id == rid), None)
3280
+ if rule is None:
3281
+ print(f"error: rule not found: {rid}", file=sys.stderr)
3282
+ return EXIT_USAGE
3283
+
3284
+ W = 66
3285
+ out = []
3286
+ out.append("=" * W)
3287
+ out.append(f"RULE: {rule.id} - {rule.name}")
3288
+ out.append(f"VERSION: {rule.rule_version}")
3289
+ out.append(f"SEVERITY: {rule.severity.value}")
3290
+ out.append(f"CATEGORY: {rule.category}")
3291
+ out.append(f"CONFIDENCE: {rule.confidence.value}")
3292
+ out.append(f"ACTION: {rule.action}")
3293
+ out.append(f"LANGUAGES: {', '.join(l.value for l in rule.languages)}")
3294
+ out.append(f"DETECTION: {rule.detection.value}")
3295
+ out.append("=" * W)
3296
+ out.append("")
3297
+ if rule.description:
3298
+ out.append("DESCRIPTION")
3299
+ for line in _wrap(rule.description, W):
3300
+ out.append(f" {line}")
3301
+ out.append("")
3302
+ if rule.message:
3303
+ out.append("MESSAGE")
3304
+ for line in _wrap(rule.message, W):
3305
+ out.append(f" {line}")
3306
+ out.append("")
3307
+ if rule.fix_before or rule.fix_after:
3308
+ out.append("TYPICAL FIX")
3309
+ if rule.fix_before:
3310
+ out.append(" Bad:")
3311
+ for ln in rule.fix_before.split("\n"):
3312
+ out.append(f" {ln}")
3313
+ if rule.fix_after:
3314
+ out.append(" Good:")
3315
+ for ln in rule.fix_after.split("\n"):
3316
+ out.append(f" {ln}")
3317
+ out.append("")
3318
+ if rule.remediation:
3319
+ out.append("REMEDIATION")
3320
+ for line in _wrap(rule.remediation, W):
3321
+ out.append(f" {line}")
3322
+ out.append("")
3323
+ if rule.references:
3324
+ out.append("REFERENCES")
3325
+ for ref in rule.references:
3326
+ out.append(f" {ref}")
3327
+ out.append("")
3328
+ out.append("-" * W)
3329
+ out.append("To get a deeper explanation, paste this output into any AI chat.")
3330
+ out.append("-" * W)
3331
+ print("\n".join(out))
3332
+ return EXIT_OK
3333
+
3334
+
3335
+ def _wrap(text: str, width: int) -> list[str]:
3336
+ """Simple word wrap without external dependencies."""
3337
+ words = text.split()
3338
+ lines: list[str] = []
3339
+ cur = ""
3340
+ for w in words:
3341
+ if not cur:
3342
+ cur = w
3343
+ elif len(cur) + 1 + len(w) <= width - 2:
3344
+ cur += " " + w
3345
+ else:
3346
+ lines.append(cur)
3347
+ cur = w
3348
+ if cur:
3349
+ lines.append(cur)
3350
+ return lines or [""]
3351
+
3352
+
3353
+ def _cmd_history(argv: list[str]) -> int:
3354
+ """Show recent scans from the local history database."""
3355
+ limit = 10
3356
+ i = 0
3357
+ while i < len(argv):
3358
+ a = argv[i]
3359
+ if a == "--limit" and i + 1 < len(argv):
3360
+ try:
3361
+ limit = max(1, int(argv[i + 1]))
3362
+ except ValueError:
3363
+ print("error: --limit must be an integer", file=sys.stderr)
3364
+ return EXIT_USAGE
3365
+ i += 2
3366
+ continue
3367
+ if a.startswith("--limit="):
3368
+ try:
3369
+ limit = max(1, int(a.split("=", 1)[1]))
3370
+ except ValueError:
3371
+ print("error: --limit must be an integer", file=sys.stderr)
3372
+ return EXIT_USAGE
3373
+ i += 1
3374
+ continue
3375
+ i += 1
3376
+
3377
+ conn = _open_history_db()
3378
+ if conn is None:
3379
+ print("error: could not open history database", file=sys.stderr)
3380
+ return EXIT_INTERNAL
3381
+ try:
3382
+ rows = conn.execute(
3383
+ """SELECT scan_id, timestamp, files_scanned, findings_count,
3384
+ critical_count, high_count, medium_count, low_count,
3385
+ commit_sha
3386
+ FROM scans
3387
+ ORDER BY timestamp DESC
3388
+ LIMIT ?""",
3389
+ (limit,),
3390
+ ).fetchall()
3391
+ except sqlite3.Error as e:
3392
+ conn.close()
3393
+ print(f"error: query failed: {e}", file=sys.stderr)
3394
+ return EXIT_INTERNAL
3395
+ conn.close()
3396
+
3397
+ if not rows:
3398
+ print("No scans recorded. Run with --history to record.")
3399
+ return EXIT_OK
3400
+
3401
+ W = 78
3402
+ print("=" * W)
3403
+ print(f"HISTORY - last {len(rows)} scan(s)")
3404
+ print(f"DB: {_history_path()}")
3405
+ print("=" * W)
3406
+ print(f"{'timestamp':<22s} {'scan_id':<10s} {'files':>5s} "
3407
+ f"{'find':>5s} {'sev (C/H/M/L)':<20s} commit")
3408
+ print("-" * W)
3409
+ for r in rows:
3410
+ (sid, ts, files, fnd, c, h, m, l, csha) = r
3411
+ ts_short = ts[:19].replace("T", " ")
3412
+ sev = f"{c}/{h}/{m}/{l}"
3413
+ csha_short = (csha[:7] if csha else "-")
3414
+ print(f"{ts_short:<22s} {sid[:8]:<10s} {files:>5d} "
3415
+ f"{fnd:>5d} {sev:<20s} {csha_short}")
3416
+ print("=" * W)
3417
+ return EXIT_OK
3418
+
3419
+
3420
+ def _cmd_stats(argv: list[str]) -> int:
3421
+ """Print summary statistics across all recorded scans."""
3422
+ conn = _open_history_db()
3423
+ if conn is None:
3424
+ print("error: could not open history database", file=sys.stderr)
3425
+ return EXIT_INTERNAL
3426
+ try:
3427
+ total = conn.execute("SELECT COUNT(*) FROM scans").fetchone()[0]
3428
+ if total == 0:
3429
+ conn.close()
3430
+ print("No scans recorded. Run with --history to record.")
3431
+ return EXIT_OK
3432
+ first_ts = conn.execute(
3433
+ "SELECT MIN(timestamp) FROM scans"
3434
+ ).fetchone()[0]
3435
+ last_ts = conn.execute(
3436
+ "SELECT MAX(timestamp) FROM scans"
3437
+ ).fetchone()[0]
3438
+ total_findings = conn.execute(
3439
+ "SELECT COALESCE(SUM(findings_count), 0) FROM scans"
3440
+ ).fetchone()[0]
3441
+ top_rules = conn.execute(
3442
+ """SELECT rule_id, COUNT(*) AS n
3443
+ FROM findings
3444
+ GROUP BY rule_id
3445
+ ORDER BY n DESC, rule_id ASC
3446
+ LIMIT 5"""
3447
+ ).fetchall()
3448
+ top_sev = conn.execute(
3449
+ """SELECT severity, COUNT(*) AS n
3450
+ FROM findings
3451
+ GROUP BY severity
3452
+ ORDER BY n DESC"""
3453
+ ).fetchall()
3454
+ except sqlite3.Error as e:
3455
+ conn.close()
3456
+ print(f"error: query failed: {e}", file=sys.stderr)
3457
+ return EXIT_INTERNAL
3458
+ conn.close()
3459
+
3460
+ W = 66
3461
+ print("=" * W)
3462
+ print("STATS")
3463
+ print(f"DB: {_history_path()}")
3464
+ print("=" * W)
3465
+ print(f" Total scans: {total}")
3466
+ print(f" Total findings: {total_findings}")
3467
+ print(f" First scan: {first_ts[:19].replace('T', ' ')}")
3468
+ print(f" Last scan: {last_ts[:19].replace('T', ' ')}")
3469
+ if total > 0:
3470
+ print(f" Avg findings/scan: {total_findings / total:.1f}")
3471
+ print()
3472
+ if top_rules:
3473
+ print(" Top rules by frequency:")
3474
+ for rid, n in top_rules:
3475
+ print(f" {rid:<10s} {n}")
3476
+ print()
3477
+ if top_sev:
3478
+ print(" Severity distribution:")
3479
+ for sev, n in top_sev:
3480
+ print(f" {sev:<10s} {n}")
3481
+ print("=" * W)
3482
+ return EXIT_OK
3483
+
3484
+
3485
+ GITHUB_ACTION_PATH = ".github/workflows/codefence.yml"
3486
+ GITHUB_ACTION_MARKER = "# Installed by: codefence init-github"
3487
+
3488
+ GITHUB_ACTION_CONTENT = """# Installed by: codefence init-github
3489
+ # Uninstall: delete this file.
3490
+ name: CodeFence
3491
+
3492
+ on:
3493
+ push:
3494
+ branches: [ "**" ]
3495
+ pull_request:
3496
+ branches: [ "**" ]
3497
+
3498
+ permissions:
3499
+ contents: read
3500
+ security-events: write
3501
+
3502
+ jobs:
3503
+ codefence:
3504
+ runs-on: ubuntu-latest
3505
+ steps:
3506
+ - name: Checkout repository
3507
+ uses: actions/checkout@v4
3508
+ with:
3509
+ fetch-depth: 0
3510
+
3511
+ - name: Setup Python
3512
+ uses: actions/setup-python@v5
3513
+ with:
3514
+ python-version: "3.12"
3515
+
3516
+ - name: Obtain CodeFence
3517
+ run: |
3518
+ set -e
3519
+ if pip install --quiet codefence 2>/dev/null; then
3520
+ echo "installed from PyPI"
3521
+ cp "$(python3 -c 'import codefence, os; print(os.path.dirname(codefence.__file__))')/codefence.py" ./codefence.py 2>/dev/null || true
3522
+ fi
3523
+ if [ ! -f codefence.py ]; then
3524
+ echo "falling back to standalone download"
3525
+ curl -fsSL -o codefence.py "https://raw.githubusercontent.com/OWNER/codefence/main/codefence.py"
3526
+ curl -fsSL -o rules.json "https://raw.githubusercontent.com/OWNER/codefence/main/rules.json"
3527
+ fi
3528
+
3529
+ - name: Run CodeFence
3530
+ run: |
3531
+ python3 codefence.py \
3532
+ --rules rules.json \
3533
+ --format sarif \
3534
+ --output results.sarif \
3535
+ --severity low \
3536
+ .
3537
+ continue-on-error: true
3538
+
3539
+ - name: Upload SARIF
3540
+ if: always()
3541
+ uses: github/codeql-action/upload-sarif@v3
3542
+ with:
3543
+ sarif_file: results.sarif
3544
+ category: codefence
3545
+
3546
+ - name: Enforce result
3547
+ run: |
3548
+ python3 codefence.py --rules rules.json --severity medium .
3549
+ """
3550
+
3551
+
3552
+ def _cmd_init_github(argv: list[str]) -> int:
3553
+ """Install a GitHub Actions workflow that runs CodeFence on every push/PR."""
3554
+ git_root = _find_git_root()
3555
+ if git_root is None:
3556
+ print("error: not in a git repository", file=sys.stderr)
3557
+ return EXIT_USAGE
3558
+
3559
+ workflow_path = git_root / GITHUB_ACTION_PATH
3560
+ if workflow_path.exists():
3561
+ content = workflow_path.read_text(encoding="utf-8", errors="replace")
3562
+ if GITHUB_ACTION_MARKER in content:
3563
+ print(f"already installed: {workflow_path}")
3564
+ return EXIT_OK
3565
+ print(f"error: {workflow_path} already exists and was not created "
3566
+ f"by CodeFence; refusing to overwrite", file=sys.stderr)
3567
+ return EXIT_USAGE
3568
+
3569
+ workflow_path.parent.mkdir(parents=True, exist_ok=True)
3570
+ workflow_path.write_text(GITHUB_ACTION_CONTENT, encoding="utf-8")
3571
+ print(f"installed: {workflow_path}")
3572
+ print()
3573
+ print("Next steps:")
3574
+ print(" 1. Edit the workflow to replace OWNER with your GitHub username")
3575
+ print(" if you are not publishing to PyPI.")
3576
+ print(" 2. Commit the file. CodeFence will run on the next push/PR.")
3577
+ return EXIT_OK
3578
+
3579
+
3580
+ def _cmd_init(argv: list[str]) -> int:
3581
+ """One-command project setup: config + hook (optional: github, baseline)."""
3582
+ with_github = "--with-github" in argv
3583
+ with_baseline = "--with-baseline" in argv
3584
+
3585
+ git_root = _find_git_root()
3586
+ if git_root is None:
3587
+ print("error: not in a git repository", file=sys.stderr)
3588
+ print(" run 'git init' first", file=sys.stderr)
3589
+ return EXIT_USAGE
3590
+
3591
+ print(f"CodeFence setup in: {git_root}")
3592
+ print()
3593
+
3594
+ # 1. Config file
3595
+ codefence_dir = git_root / BASELINE_DIR
3596
+ config_path = codefence_dir / CONFIG_FILE
3597
+ if config_path.exists():
3598
+ print(f"[skip] config already exists: {config_path}")
3599
+ else:
3600
+ codefence_dir.mkdir(parents=True, exist_ok=True)
3601
+ config_path.write_text(
3602
+ json.dumps(DEFAULT_CONFIG_CONTENT, indent=2, ensure_ascii=False),
3603
+ encoding="utf-8",
3604
+ )
3605
+ print(f"[new] config: {config_path}")
3606
+
3607
+ # 2. Pre-commit hook
3608
+ rc = _cmd_init_hook([])
3609
+ if rc != EXIT_OK:
3610
+ return rc
3611
+
3612
+ # 3. Optional: GitHub Action
3613
+ if with_github:
3614
+ rc = _cmd_init_github([])
3615
+ if rc != EXIT_OK:
3616
+ return rc
3617
+
3618
+ # 4. Optional: baseline
3619
+ if with_baseline:
3620
+ rc = _cmd_baseline([])
3621
+ if rc != EXIT_OK:
3622
+ return rc
3623
+
3624
+ print()
3625
+ print("Setup complete. Try:")
3626
+ print(" codefence --staged # scan only what is staged")
3627
+ print(" codefence --staged --diff # only NEW findings")
3628
+ print(" git commit # hook runs automatically")
3629
+ return EXIT_OK
3630
+
3631
+
3632
+ def _cmd_policy(argv: list[str]) -> int:
3633
+ if not argv or argv[0] != "validate":
3634
+ print("usage: codefence policy validate FILE",
3635
+ file=sys.stderr)
3636
+ return EXIT_USAGE
3637
+ if len(argv) < 2:
3638
+ print("usage: codefence policy validate FILE",
3639
+ file=sys.stderr)
3640
+ return EXIT_USAGE
3641
+ path = Path(argv[1])
3642
+ try:
3643
+ data = _load_policy(path)
3644
+ except (OSError, ValueError, json.JSONDecodeError) as e:
3645
+ print(f"error: {e}", file=sys.stderr)
3646
+ return EXIT_INTERNAL
3647
+ n_overrides = len(data.get("overrides", {}) or {})
3648
+ n_extra = len(data.get("rules", []) or [])
3649
+ print(f"policy valid: {path}")
3650
+ print(f" overrides: {n_overrides}")
3651
+ print(f" extra rules: {n_extra}")
3652
+ return EXIT_OK
3653
+
3654
+
3655
+ SUBCOMMANDS = {
3656
+ "init-hook", "uninstall-hook", "init", "baseline",
3657
+ "check", "explain", "policy", "history", "stats",
3658
+ "init-github",
3659
+ }
3660
+
3661
+ HOOK_MARKER = "# CodeFence pre-commit hook"
3662
+ HOOK_BACKUP_SUFFIX = ".pre-codefence"
3663
+ HOOK_CONTENT = """#!/bin/sh
3664
+ # CodeFence pre-commit hook
3665
+ # Installed by: codefence init-hook
3666
+ # To uninstall: codefence uninstall-hook
3667
+
3668
+ # Environment override (optional):
3669
+ # CODEFENCE_CMD - explicit command or path (e.g. "python3 /path/to/codefence.py")
3670
+ # CODEFENCE_RULES - explicit rules.json path
3671
+
3672
+ if [ -n "$CODEFENCE_CMD" ]; then
3673
+ CF="$CODEFENCE_CMD"
3674
+ elif command -v cfence >/dev/null 2>&1; then
3675
+ CF=cfence
3676
+ elif command -v codefence >/dev/null 2>&1; then
3677
+ CF=codefence
3678
+ elif [ -f "./codefence.py" ]; then
3679
+ CF="python3 ./codefence.py"
3680
+ elif [ -f "./cfence.py" ]; then
3681
+ CF="python3 ./cfence.py"
3682
+ else
3683
+ echo "[codefence] executable not found in PATH or current directory"
3684
+ exit 0
3685
+ fi
3686
+
3687
+ if [ -n "$CODEFENCE_RULES" ]; then
3688
+ RULES="$CODEFENCE_RULES"
3689
+ elif [ -f "rules.json" ]; then
3690
+ RULES="rules.json"
3691
+ else
3692
+ RULES="$(dirname "$0")/../../rules.json"
3693
+ fi
3694
+
3695
+ $CF --rules "$RULES" --staged
3696
+ exit $?
3697
+ """
3698
+
3699
+
3700
+ def _find_git_root(start: Path | None = None) -> Path | None:
3701
+ p = (start or Path.cwd()).resolve()
3702
+ while True:
3703
+ if (p / ".git").is_dir():
3704
+ return p
3705
+ if p.parent == p:
3706
+ return None
3707
+ p = p.parent
3708
+
3709
+
3710
+ def _get_staged_files() -> list[Path] | None:
3711
+ """Return list of staged file paths, or None if git unavailable/not a repo."""
3712
+ try:
3713
+ result = subprocess.run(
3714
+ ["git", "diff", "--cached", "--name-only", "--diff-filter=ACMR"],
3715
+ capture_output=True, text=True, check=False,
3716
+ )
3717
+ except (FileNotFoundError, OSError):
3718
+ return None
3719
+ if result.returncode != 0:
3720
+ return None
3721
+ return [Path(line.strip()) for line in result.stdout.splitlines()
3722
+ if line.strip()]
3723
+
3724
+
3725
+ def _cmd_baseline(argv: list[str]) -> int:
3726
+ """Create .codefence/baseline.json from the current scan."""
3727
+ git_root = _find_git_root()
3728
+ if git_root is None:
3729
+ print("error: not in a git repository", file=sys.stderr)
3730
+ return EXIT_USAGE
3731
+ # Parse minimal argv for --rules / paths / etc, but keep it simple:
3732
+ parser = _build_argparser()
3733
+ args = parser.parse_args(argv)
3734
+ try:
3735
+ cfg = _merge_config(args)
3736
+ except SystemExit as e:
3737
+ print(str(e), file=sys.stderr)
3738
+ return EXIT_USAGE
3739
+
3740
+ rules_path = Path(cfg.rules_file) if cfg.rules_file else _default_rules_path()
3741
+ if not rules_path.is_file():
3742
+ print(f"error: rules file not found: {rules_path}", file=sys.stderr)
3743
+ return EXIT_USAGE
3744
+ try:
3745
+ rules = load_rules(rules_path)
3746
+ except (OSError, ValueError, json.JSONDecodeError) as e:
3747
+ print(f"error: failed to load rules: {e}", file=sys.stderr)
3748
+ return EXIT_INTERNAL
3749
+
3750
+ targets = _collect_paths(args.paths, cfg.include, cfg.exclude)
3751
+ if not targets:
3752
+ targets = _collect_paths(["."], cfg.include, cfg.exclude)
3753
+
3754
+ findings: list[Finding] = []
3755
+ for target in targets:
3756
+ findings.extend(scan_file(target, rules, cfg.max_size))
3757
+ findings.sort(key=Finding.sort_key)
3758
+
3759
+ baseline_path = _baseline_path(git_root)
3760
+ _save_baseline(baseline_path, findings)
3761
+ print(f"baseline written: {baseline_path}")
3762
+ print(f" {len(findings)} finding(s) recorded")
3763
+ return EXIT_OK
3764
+
3765
+
3766
+ def _cmd_init_hook(argv: list[str]) -> int:
3767
+ git_root = _find_git_root()
3768
+ if git_root is None:
3769
+ print("error: not in a git repository", file=sys.stderr)
3770
+ return EXIT_USAGE
3771
+ hooks_dir = git_root / ".git" / "hooks"
3772
+ hooks_dir.mkdir(exist_ok=True)
3773
+ hook_path = hooks_dir / "pre-commit"
3774
+
3775
+ if hook_path.exists():
3776
+ content = hook_path.read_text(encoding="utf-8", errors="replace")
3777
+ if HOOK_MARKER in content:
3778
+ print(f"already installed: {hook_path}")
3779
+ return EXIT_OK
3780
+ backup = Path(str(hook_path) + HOOK_BACKUP_SUFFIX)
3781
+ if not backup.exists():
3782
+ backup.write_text(content, encoding="utf-8")
3783
+ print(f"backed up existing hook to: {backup}")
3784
+
3785
+ hook_path.write_text(HOOK_CONTENT, encoding="utf-8")
3786
+ try:
3787
+ os.chmod(hook_path, 0o755)
3788
+ except OSError:
3789
+ pass
3790
+ print(f"installed: {hook_path}")
3791
+ return EXIT_OK
3792
+
3793
+
3794
+ def _cmd_uninstall_hook(argv: list[str]) -> int:
3795
+ git_root = _find_git_root()
3796
+ if git_root is None:
3797
+ print("error: not in a git repository", file=sys.stderr)
3798
+ return EXIT_USAGE
3799
+ hook_path = git_root / ".git" / "hooks" / "pre-commit"
3800
+ if not hook_path.exists():
3801
+ print("no pre-commit hook found")
3802
+ return EXIT_OK
3803
+ content = hook_path.read_text(encoding="utf-8", errors="replace")
3804
+ if HOOK_MARKER not in content:
3805
+ print("pre-commit hook exists but was not installed by CodeFence; "
3806
+ "refusing to remove", file=sys.stderr)
3807
+ return EXIT_USAGE
3808
+ backup = Path(str(hook_path) + HOOK_BACKUP_SUFFIX)
3809
+ if backup.exists():
3810
+ hook_path.write_text(backup.read_text(encoding="utf-8"),
3811
+ encoding="utf-8")
3812
+ backup.unlink()
3813
+ print("restored previous pre-commit hook")
3814
+ else:
3815
+ hook_path.unlink()
3816
+ print("removed CodeFence pre-commit hook")
3817
+ return EXIT_OK
3818
+
3819
+
3820
+ DEFAULT_EXCLUDES = (
3821
+ "node_modules", ".git", "venv", ".venv", "__pycache__",
3822
+ "dist", "build", ".tox", ".mypy_cache", ".pytest_cache",
3823
+ ".idea", ".vscode", "target", "vendor",
3824
+ )
3825
+
3826
+ DEFAULT_INCLUDES = ("*.py", "*.js", "*.mjs", "*.cjs")
3827
+
3828
+
3829
+ @dataclass(frozen=True, slots=True)
3830
+ class ScanConfig:
3831
+ include: tuple[str, ...] = DEFAULT_INCLUDES
3832
+ exclude: tuple[str, ...] = DEFAULT_EXCLUDES
3833
+ max_size: int = DEFAULT_MAX_SIZE
3834
+ severity: str = "low"
3835
+ format: str = "cli"
3836
+ cache: bool = False
3837
+ rules_file: str | None = None
3838
+
3839
+
3840
+ def _load_config_file(path: Path) -> dict:
3841
+ if not path.is_file():
3842
+ raise FileNotFoundError(f"config file not found: {path}")
3843
+ data = json.loads(path.read_text(encoding="utf-8"))
3844
+ if not isinstance(data, dict):
3845
+ raise ValueError("config file must contain a JSON object")
3846
+ return data
3847
+
3848
+
3849
+ def _merge_config(args: argparse.Namespace) -> ScanConfig:
3850
+ """Priority: CLI args > config file > built-in defaults."""
3851
+ file_cfg: dict = {}
3852
+ config_source: Path | None = None
3853
+ if args.config:
3854
+ config_source = Path(args.config)
3855
+ else:
3856
+ git_root = _find_git_root()
3857
+ if git_root is not None:
3858
+ auto = git_root / BASELINE_DIR / CONFIG_FILE
3859
+ if auto.is_file():
3860
+ config_source = auto
3861
+ if config_source is not None:
3862
+ try:
3863
+ file_cfg = _load_config_file(config_source)
3864
+ except (OSError, ValueError, json.JSONDecodeError) as e:
3865
+ raise SystemExit(f"error: failed to load config: {e}")
3866
+
3867
+ def pick(cli_value, key, default):
3868
+ if cli_value is not None:
3869
+ return cli_value
3870
+ if key in file_cfg:
3871
+ return file_cfg[key]
3872
+ return default
3873
+
3874
+ include = pick(args.include, "include", DEFAULT_INCLUDES)
3875
+ exclude = pick(args.exclude, "exclude", DEFAULT_EXCLUDES)
3876
+ max_size = pick(args.max_size, "max_size", DEFAULT_MAX_SIZE)
3877
+ severity = pick(args.severity_cli, "severity", "low")
3878
+ fmt = pick(args.format_cli, "format", "cli")
3879
+ cache = pick(args.cache, "cache", False)
3880
+ rules_file = pick(args.rules, "rules", None)
3881
+
3882
+ if isinstance(include, str):
3883
+ include = (include,)
3884
+ if isinstance(exclude, str):
3885
+ exclude = (exclude,)
3886
+
3887
+ return ScanConfig(
3888
+ include=tuple(include),
3889
+ exclude=tuple(exclude),
3890
+ max_size=int(max_size),
3891
+ severity=str(severity),
3892
+ format=str(fmt),
3893
+ cache=bool(cache),
3894
+ rules_file=rules_file if rules_file else None,
3895
+ )
3896
+
3897
+
3898
+ def _is_excluded(path: Path, excludes: Sequence[str]) -> bool:
3899
+ parts = set(path.parts)
3900
+ for ex in excludes:
3901
+ if ex in parts:
3902
+ return True
3903
+ return False
3904
+
3905
+
3906
+ def _matches_include(path: Path, includes: Sequence[str]) -> bool:
3907
+ name = path.name
3908
+ for pat in includes:
3909
+ # simple glob: "*.py", "*.js", or exact
3910
+ if pat.startswith("*.") and name.endswith(pat[1:]):
3911
+ return True
3912
+ if pat == name:
3913
+ return True
3914
+ return False
3915
+
3916
+
3917
+ def _build_argparser() -> argparse.ArgumentParser:
3918
+ p = argparse.ArgumentParser(
3919
+ prog=TOOL_NAME,
3920
+ description="Offline AI code sanity check (pattern-based).",
3921
+ add_help=False,
3922
+ )
3923
+ p.add_argument("-h", "--help", action="store_true", dest="show_help",
3924
+ help="Show full help and exit.")
3925
+ p.add_argument("paths", nargs="*",
3926
+ help="Files or directories to scan.")
3927
+ p.add_argument("--rules", default=None,
3928
+ help="Path to rules.json.")
3929
+
3930
+ p.add_argument("--max-size", type=int, default=None)
3931
+ p.add_argument("--format-cli", dest="format_cli", default=None,
3932
+ choices=["cli", "json", "html", "sarif"])
3933
+ p.add_argument("--format", dest="format_cli", default=None,
3934
+ choices=["cli", "json", "html", "sarif"])
3935
+ p.add_argument("--output", default=None,
3936
+ help="Write report to file instead of stdout.")
3937
+ p.add_argument("--include", action="append", default=None,
3938
+ help="Glob of files to include (repeatable).")
3939
+ p.add_argument("--exclude", action="append", default=None,
3940
+ help="Directory or file name to exclude (repeatable).")
3941
+ p.add_argument("--config", default=None,
3942
+ help="Path to a JSON config file.")
3943
+ p.add_argument("--cache", action="store_true", default=None,
3944
+ help="Enable local cache (off by default).")
3945
+ p.add_argument("--staged", action="store_true", default=False,
3946
+ help="Scan only files currently staged in git.")
3947
+ p.add_argument("--diff", action="store_true", default=False,
3948
+ help="Report only findings NOT in the baseline.")
3949
+ p.add_argument("--no-baseline", action="store_true", default=False,
3950
+ help="Ignore the baseline even if --diff is set.")
3951
+ p.add_argument("--policy", default=None,
3952
+ help="Path to a policy JSON file.")
3953
+ p.add_argument("--evidence", default=None,
3954
+ help="Write a deterministic evidence JSON file.")
3955
+ p.add_argument("--history", action="store_true", default=False,
3956
+ help="Record this scan in the local history DB.")
3957
+ p.add_argument("--no-color", action="store_true",
3958
+ help="Disable ANSI colors.")
3959
+ p.add_argument("-q", "--quiet", action="store_true",
3960
+ help="Only print summary.")
3961
+ p.add_argument("--version", action="version",
3962
+ version=f"{TOOL_NAME} {TOOL_VERSION}")
3963
+ p.add_argument("--severity", dest="severity_cli", default=None,
3964
+ choices=["critical", "high", "medium", "low", "info"])
3965
+ return p
3966
+
3967
+
3968
+ def _default_rules_path() -> Path:
3969
+ return Path(__file__).resolve().parent / DEFAULT_RULES_FILENAME
3970
+
3971
+
3972
+ def _collect_paths(paths: Sequence[str],
3973
+ includes: Sequence[str] = DEFAULT_INCLUDES,
3974
+ excludes: Sequence[str] = DEFAULT_EXCLUDES) -> list[Path]:
3975
+ out: list[Path] = []
3976
+ seen: set[str] = set()
3977
+ for raw in paths:
3978
+ p = Path(raw)
3979
+ if p.is_file():
3980
+ if _matches_include(p, includes):
3981
+ key = str(p)
3982
+ if key not in seen:
3983
+ seen.add(key)
3984
+ out.append(p)
3985
+ continue
3986
+ if not p.is_dir():
3987
+ continue
3988
+ for child in p.rglob("*"):
3989
+ if not child.is_file():
3990
+ continue
3991
+ if _is_excluded(child, excludes):
3992
+ continue
3993
+ if not _matches_include(child, includes):
3994
+ continue
3995
+ key = str(child)
3996
+ if key in seen:
3997
+ continue
3998
+ seen.add(key)
3999
+ out.append(child)
4000
+ return sorted(out)
4001
+
4002
+
4003
+ def _print_welcome() -> None:
4004
+ use_color = _color_enabled(False)
4005
+
4006
+ def c(text: str, sev: Severity = Severity.INFO) -> str:
4007
+ return _c(text, sev, use_color)
4008
+
4009
+ W = 64
4010
+ out: list[str] = []
4011
+ out.append(_box_top(W))
4012
+ out.append(_box_row("", W))
4013
+ out.append(_box_row(
4014
+ f" {c('CodeFence')} {c(chr(0x00b7), Severity.LOW)} "
4015
+ f"v{TOOL_VERSION}", W))
4016
+ out.append(_box_row(" Offline AI code sanity check", W))
4017
+ out.append(_box_row("", W))
4018
+ out.append(_box_bot(W))
4019
+ out.append("")
4020
+ out.append(" Pattern-based. Zero network calls. Python + JavaScript.")
4021
+ out.append("")
4022
+ out.append(f" {c('QUICK START', Severity.LOW)}")
4023
+ out.append(f" {TOOL_NAME}.py <file-or-directory>")
4024
+ out.append("")
4025
+ out.append(f" {c('EXAMPLES', Severity.LOW)}")
4026
+ out.append(f" {TOOL_NAME}.py src/")
4027
+ out.append(f" {TOOL_NAME}.py --format json app.py")
4028
+ out.append(f" {TOOL_NAME}.py --format html "
4029
+ f"--output report.html src/")
4030
+ out.append(f" {TOOL_NAME}.py --severity high .")
4031
+ out.append("")
4032
+ out.append(f" {c('COMMON OPTIONS', Severity.LOW)}")
4033
+ out.append(" --format {cli,json,html,sarif} "
4034
+ "Output format (default: cli)")
4035
+ out.append(" --output FILE "
4036
+ "Write report to file")
4037
+ out.append(" --severity LEVEL "
4038
+ "Minimum severity (default: low)")
4039
+ out.append(" --include GLOB "
4040
+ "Files to include (repeatable)")
4041
+ out.append(" --exclude NAME "
4042
+ "Directory/file to skip (repeatable)")
4043
+ out.append(" -q, --quiet Summary only")
4044
+ out.append("")
4045
+ out.append(f" Full help: {TOOL_NAME}.py --help")
4046
+ out.append("")
4047
+ print("\n".join(out))
4048
+
4049
+
4050
+ def _print_help() -> None:
4051
+ use_color = _color_enabled(False)
4052
+
4053
+ def h(text: str) -> str:
4054
+ return _c(text, Severity.INFO, use_color)
4055
+
4056
+ def s(text: str) -> str:
4057
+ return _c(text, Severity.LOW, use_color)
4058
+
4059
+ blocks: list[str] = []
4060
+ blocks.append(f"{h('CodeFence')} "
4061
+ f"{s(chr(0x00b7))} v{TOOL_VERSION}")
4062
+ blocks.append("Offline AI code sanity check. Pattern-based, "
4063
+ "not a security audit.")
4064
+ blocks.append("")
4065
+ blocks.append(h("USAGE"))
4066
+ blocks.append(f" cfence [OPTIONS] PATH...")
4067
+ blocks.append(f" {TOOL_NAME}.py [OPTIONS] PATH... (source checkout)")
4068
+ blocks.append("")
4069
+ blocks.append(h("DESCRIPTION"))
4070
+ blocks.append(
4071
+ " Scans AI-generated Python and JavaScript source files for")
4072
+ blocks.append(
4073
+ " selected dangerous patterns (hardcoded secrets, injection,")
4074
+ blocks.append(
4075
+ " weak cryptography, unsafe config) before you commit.")
4076
+ blocks.append(
4077
+ " Fully offline. Zero network calls. Zero telemetry.")
4078
+ blocks.append("")
4079
+ blocks.append(h("OUTPUT FORMATS"))
4080
+ blocks.append(" cli Colored terminal report (default)")
4081
+ blocks.append(" json Machine-readable JSON for automation")
4082
+ blocks.append(" html Self-contained HTML report "
4083
+ "(dark/light theme)")
4084
+ blocks.append(" sarif SARIF 2.1.0 for GitHub Code Scanning")
4085
+ blocks.append("")
4086
+ blocks.append(h("OPTIONS"))
4087
+ options = [
4088
+ ("--format FORMAT",
4089
+ "Output format: cli, json, html, sarif (default: cli)"),
4090
+ ("--output FILE",
4091
+ "Write report to file instead of stdout"),
4092
+ ("--severity LEVEL",
4093
+ "Minimum severity to report: critical, high, medium, low, "
4094
+ "info (default: low)"),
4095
+ ("--include GLOB",
4096
+ "Glob of files to include (repeatable). "
4097
+ "Default: *.py *.js *.mjs *.cjs"),
4098
+ ("--exclude NAME",
4099
+ "Directory or file name to skip (repeatable). "
4100
+ "Default: node_modules, .git, venv, __pycache__, ..."),
4101
+ ("--config FILE",
4102
+ "Load options from a JSON config file"),
4103
+ ("--max-size BYTES",
4104
+ f"Skip files larger than N bytes (default: {DEFAULT_MAX_SIZE})"),
4105
+ ("--rules FILE",
4106
+ "Path to rules.json (default: sibling of this script)"),
4107
+ ("--cache",
4108
+ "Enable local cache (off by default)"),
4109
+ ("--no-color", "Disable ANSI colors"),
4110
+ ("-q, --quiet", "Print only the summary"),
4111
+ ("-h, --help", "Show this help and exit"),
4112
+ ("--version", "Print version and exit"),
4113
+ ]
4114
+ for flag, desc in options:
4115
+ blocks.append(f" {flag:<24s} {desc}")
4116
+ blocks.append("")
4117
+ blocks.append(h("EXAMPLES"))
4118
+ examples = [
4119
+ f"{TOOL_NAME}.py app.py",
4120
+ f"{TOOL_NAME}.py src/",
4121
+ f"{TOOL_NAME}.py --format json --output report.json src/",
4122
+ f"{TOOL_NAME}.py --format sarif -o results.sarif .",
4123
+ f"{TOOL_NAME}.py --severity high --exclude tests .",
4124
+ f"{TOOL_NAME}.py --config sanitizer.json .",
4125
+ ]
4126
+ for ex in examples:
4127
+ blocks.append(f" #")
4128
+ blocks.append(f" {ex}")
4129
+ blocks.append("")
4130
+ blocks.append(h("EXIT CODES"))
4131
+ blocks.append(" 0 No findings "
4132
+ "(or all below the severity threshold)")
4133
+ blocks.append(" 1 At least one finding at or above "
4134
+ "the threshold")
4135
+ blocks.append(" 2 Usage error (invalid arguments)")
4136
+ blocks.append(" 3 Internal error")
4137
+ blocks.append("")
4138
+ blocks.append(h("USING IN CI/CD AND AGENT PIPELINES"))
4139
+ blocks.append(" - JSON and SARIF are pure on stdout "
4140
+ "(no banner, no ANSI).")
4141
+ blocks.append(" - Exit codes are stable and documented above.")
4142
+ blocks.append(" - Safe for headless use: no stdin, no prompts.")
4143
+ blocks.append(" - Never performs network calls.")
4144
+ blocks.append("")
4145
+ print("\n".join(blocks))
4146
+
4147
+
4148
+ def main(argv: Sequence[str] | None = None) -> int:
4149
+ if argv is None:
4150
+ argv = sys.argv[1:]
4151
+
4152
+ # No arguments at all → friendly welcome screen.
4153
+ if not argv:
4154
+ _print_welcome()
4155
+ return EXIT_OK
4156
+
4157
+ # Subcommand dispatch (allows flags before subcommand)
4158
+ sub_idx, sub = _find_subcommand(argv)
4159
+ if sub is not None:
4160
+ rest = argv[:sub_idx] + argv[sub_idx + 1:]
4161
+ if sub == "init-hook":
4162
+ return _cmd_init_hook(rest)
4163
+ if sub == "uninstall-hook":
4164
+ return _cmd_uninstall_hook(rest)
4165
+ if sub == "baseline":
4166
+ return _cmd_baseline(rest)
4167
+ if sub == "policy":
4168
+ return _cmd_policy(rest)
4169
+ if sub == "explain":
4170
+ return _cmd_explain(rest)
4171
+ if sub == "history":
4172
+ return _cmd_history(rest)
4173
+ if sub == "stats":
4174
+ return _cmd_stats(rest)
4175
+ if sub == "init-github":
4176
+ return _cmd_init_github(rest)
4177
+ if sub == "init":
4178
+ return _cmd_init(rest)
4179
+ print(f"error: subcommand '{sub}' not implemented yet",
4180
+ file=sys.stderr)
4181
+ return EXIT_USAGE
4182
+
4183
+ parser = _build_argparser()
4184
+ args = parser.parse_args(argv)
4185
+
4186
+ if getattr(args, "show_help", False):
4187
+ _print_help()
4188
+ return EXIT_OK
4189
+
4190
+ try:
4191
+ cfg = _merge_config(args)
4192
+ except SystemExit as e:
4193
+ print(str(e), file=sys.stderr)
4194
+ return EXIT_USAGE
4195
+
4196
+ rules_path = Path(cfg.rules_file) if cfg.rules_file else _default_rules_path()
4197
+ if not rules_path.is_file():
4198
+ print(f"error: rules file not found: {rules_path}", file=sys.stderr)
4199
+ return EXIT_USAGE
4200
+ try:
4201
+ rules = load_rules(rules_path)
4202
+ except (OSError, ValueError, json.JSONDecodeError) as e:
4203
+ print(f"error: failed to load rules: {e}", file=sys.stderr)
4204
+ return EXIT_INTERNAL
4205
+
4206
+ # Apply --policy if provided
4207
+ if getattr(args, "policy", None):
4208
+ try:
4209
+ policy = _load_policy(Path(args.policy))
4210
+ rules = _apply_policy(rules, policy)
4211
+ except (OSError, ValueError, json.JSONDecodeError) as e:
4212
+ print(f"error: failed to load policy: {e}", file=sys.stderr)
4213
+ return EXIT_INTERNAL
4214
+
4215
+ # If --staged, override targets with staged files from git
4216
+ if getattr(args, "staged", False):
4217
+ staged = _get_staged_files()
4218
+ if staged is None:
4219
+ print("error: --staged requires a git repository with git "
4220
+ "installed", file=sys.stderr)
4221
+ return EXIT_USAGE
4222
+ if not staged:
4223
+ # No staged files: exit clean, print short message
4224
+ if cfg.format == "cli":
4225
+ print("No staged files.")
4226
+ return EXIT_OK
4227
+ # Warn if the policy or config file is itself being staged
4228
+ # in this commit — that allows a PR to weaken its own gate.
4229
+ staged_set = {str(t) for t in staged}
4230
+ watch_paths = []
4231
+ if getattr(args, "policy", None):
4232
+ watch_paths.append(Path(args.policy))
4233
+ git_root = _find_git_root()
4234
+ if git_root is not None:
4235
+ watch_paths.append(git_root / BASELINE_DIR / "policy.json")
4236
+ watch_paths.append(git_root / BASELINE_DIR / CONFIG_FILE)
4237
+ # Deduplicate watch paths (resolve to absolute)
4238
+ seen_wp: set[str] = set()
4239
+ unique_wp = []
4240
+ for wp in watch_paths:
4241
+ try:
4242
+ resolved = str(wp.resolve())
4243
+ except OSError:
4244
+ resolved = str(wp)
4245
+ if resolved in seen_wp:
4246
+ continue
4247
+ seen_wp.add(resolved)
4248
+ unique_wp.append(wp)
4249
+ for wp in unique_wp:
4250
+ try:
4251
+ rel = wp.relative_to(git_root) if git_root else wp
4252
+ except ValueError:
4253
+ rel = wp
4254
+ if str(rel) in staged_set or str(wp) in staged_set:
4255
+ print(
4256
+ f"[codefence] warning: {rel} is staged in this "
4257
+ f"commit. If this is a pull request, the change may "
4258
+ f"weaken or bypass the gate. Review carefully before "
4259
+ f"merging.",
4260
+ file=sys.stderr,
4261
+ )
4262
+ targets = [t for t in staged
4263
+ if _matches_include(t, cfg.include)
4264
+ and not _is_excluded(t, cfg.exclude)]
4265
+ else:
4266
+ targets = _collect_paths(args.paths, cfg.include, cfg.exclude)
4267
+ # Note: empty targets still produces output (empty report),
4268
+ # so CI/CD pipelines always get valid JSON/SARIF.
4269
+
4270
+ threshold = severity_rank(Severity(cfg.severity))
4271
+ started = time.perf_counter()
4272
+ all_findings: list[Finding] = []
4273
+ rules_fp = ""
4274
+ if cfg.cache:
4275
+ rules_fp = _rules_fingerprint(rules_path)
4276
+ for target in targets:
4277
+ all_findings.extend(scan_file(
4278
+ target, rules, cfg.max_size,
4279
+ use_cache=cfg.cache, rules_fingerprint=rules_fp,
4280
+ ))
4281
+ duration_ms = int((time.perf_counter() - started) * 1000)
4282
+
4283
+ all_findings = [f for f in all_findings
4284
+ if f.severity == Severity.INFO
4285
+ or severity_rank(f.severity) <= threshold]
4286
+ all_findings.sort(key=Finding.sort_key)
4287
+
4288
+ # --diff: filter to only new findings vs baseline
4289
+ if getattr(args, "diff", False) and not getattr(args, "no_baseline", False):
4290
+ baseline_path = _baseline_path()
4291
+ baseline_set = _load_baseline(baseline_path)
4292
+ if baseline_set is None:
4293
+ print(f"warning: baseline not found at {baseline_path}; "
4294
+ f"run 'codefence baseline' first", file=sys.stderr)
4295
+ # Fall through: show all findings (safe default)
4296
+ else:
4297
+ before = len(all_findings)
4298
+ all_findings = _filter_new_findings(all_findings, baseline_set)
4299
+ suppressed = before - len(all_findings)
4300
+ if cfg.format == "cli":
4301
+ print(f"Baseline: {len(baseline_set)} finding(s) recorded; "
4302
+ f"{suppressed} existing finding(s) suppressed.")
4303
+ print()
4304
+
4305
+ # Apply action policy:
4306
+ # allow -> remove finding entirely (not shown, not counted)
4307
+ # warn -> shown, but does NOT affect exit code
4308
+ # block -> shown AND affects exit code
4309
+ allow_suppressed = 0
4310
+ if any(getattr(f, "action", "block") != "block" for f in all_findings):
4311
+ filtered = []
4312
+ for f in all_findings:
4313
+ act = getattr(f, "action", "block")
4314
+ if act == "allow":
4315
+ allow_suppressed += 1
4316
+ continue
4317
+ filtered.append(f)
4318
+ all_findings = filtered
4319
+ if allow_suppressed and cfg.format == "cli":
4320
+ print(f"Policy: {allow_suppressed} finding(s) suppressed by 'allow' rules.")
4321
+ print()
4322
+
4323
+ fmt = cfg.format
4324
+ if fmt == "cli":
4325
+ use_color = _color_enabled(args.no_color)
4326
+ out = report_cli(all_findings, len(targets), duration_ms,
4327
+ use_color, quiet=args.quiet)
4328
+ elif fmt == "json":
4329
+ out = report_json(all_findings, len(targets), duration_ms)
4330
+ elif fmt == "html":
4331
+ out = report_html(all_findings, len(targets), duration_ms)
4332
+ elif fmt == "sarif":
4333
+ out = report_sarif(rules, all_findings)
4334
+ else:
4335
+ print(f"error: unknown format {fmt!r}", file=sys.stderr)
4336
+ return EXIT_USAGE
4337
+
4338
+ if args.output:
4339
+ try:
4340
+ Path(args.output).write_text(out, encoding="utf-8")
4341
+ except OSError as e:
4342
+ print(f"error: failed to write output: {e}", file=sys.stderr)
4343
+ return EXIT_INTERNAL
4344
+ else:
4345
+ print(out)
4346
+
4347
+ # --evidence or --history: build evidence struct
4348
+ need_evidence = (getattr(args, "evidence", None)
4349
+ or getattr(args, "history", False))
4350
+ if need_evidence:
4351
+ policy_path = Path(args.policy) if getattr(args, "policy", None) else None
4352
+ baseline_path = _baseline_path()
4353
+ if not baseline_path.is_file():
4354
+ baseline_path = None
4355
+ evidence = _build_evidence(
4356
+ findings=all_findings,
4357
+ files_scanned=len(targets),
4358
+ duration_ms=duration_ms,
4359
+ rules_path=rules_path,
4360
+ policy_path=policy_path,
4361
+ baseline_path=baseline_path,
4362
+ )
4363
+ if getattr(args, "history", False):
4364
+ _record_scan(evidence, all_findings)
4365
+ if cfg.format == "cli":
4366
+ print()
4367
+ print(f"History: scan {evidence['scan_id'][:8]} recorded.")
4368
+ if getattr(args, "evidence", None):
4369
+ try:
4370
+ Path(args.evidence).write_text(
4371
+ json.dumps(evidence, indent=2, ensure_ascii=False),
4372
+ encoding="utf-8",
4373
+ )
4374
+ except OSError as e:
4375
+ print(f"error: failed to write evidence: {e}",
4376
+ file=sys.stderr)
4377
+ return EXIT_INTERNAL
4378
+ if cfg.format == "cli":
4379
+ print()
4380
+ print(f"Evidence written: {args.evidence}")
4381
+ print(f" scan_id: {evidence['scan_id']}")
4382
+ print(f" result_hash: {evidence['result_hash'][:32]}...")
4383
+
4384
+ if not all_findings:
4385
+ return EXIT_OK
4386
+ if any(
4387
+ f.severity in (Severity.CRITICAL, Severity.HIGH,
4388
+ Severity.MEDIUM, Severity.LOW)
4389
+ and getattr(f, "action", "block") == "block"
4390
+ for f in all_findings
4391
+ ):
4392
+ return EXIT_FINDINGS
4393
+ return EXIT_OK
4394
+
4395
+
4396
+ if __name__ == "__main__":
4397
+ sys.exit(main())