oat-py 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
oat/__init__.py ADDED
@@ -0,0 +1,4 @@
1
+ # OAT - OpenHarmony OSS Audit Tool (Python Edition)
2
+ # Copyright (c) 2024 Huawei Device Co., Ltd.
3
+ # Licensed under the Apache License, Version 2.0
4
+ __version__ = "2.0.0"
oat/__main__.py ADDED
@@ -0,0 +1,40 @@
1
+ """
2
+ Entry point: python -m oat or oat (if installed via pip)
3
+ """
4
+ import sys
5
+ import time
6
+ from pathlib import Path
7
+
8
+
9
+ def main(argv=None):
10
+ from oat.cli import parse_args
11
+ args = parse_args(argv)
12
+
13
+ start = time.time()
14
+ print(f"[OAT] OpenHarmony OSS Audit Tool - Python Edition")
15
+ print(f"[OAT] Scanning: {args.s}")
16
+
17
+ from oat.config.loader import build_task
18
+ from oat.analysis.pipeline import run_pipeline
19
+ from oat.reporter.plain_reporter import PlainReporter
20
+
21
+ task = build_task(args)
22
+
23
+ report_model = run_pipeline(task)
24
+
25
+ reporter = PlainReporter(task, report_model)
26
+ reporter.write()
27
+
28
+ elapsed = time.time() - start
29
+ total = report_model.total_issues
30
+ print(f"[OAT] Scan complete in {elapsed:.1f}s. Issues found: {total}")
31
+ if total > 0:
32
+ print(f"[OAT] Report: {task.report_dir}")
33
+ sys.exit(1)
34
+ else:
35
+ print("[OAT] No compliance issues found.")
36
+ sys.exit(0)
37
+
38
+
39
+ if __name__ == "__main__":
40
+ main()
File without changes
@@ -0,0 +1,36 @@
1
+ """
2
+ FileDocument — mirrors OatFileDocument.java.
3
+ Holds analysis results for a single file.
4
+ """
5
+ from __future__ import annotations
6
+
7
+ from dataclasses import dataclass, field
8
+ from pathlib import Path
9
+ from typing import List, Optional
10
+
11
+
12
+ @dataclass
13
+ class FileDocument:
14
+ path: Path
15
+ rel_path: str = "" # relative to src_dir, forward slashes
16
+
17
+ # Set by file_type analyser
18
+ file_type: str = "text" # 'text' | 'binary' | 'archive'
19
+
20
+ # Set by header analyser
21
+ license: str = "NoLicenseHeader"
22
+ copyright_owners: List[str] = field(default_factory=list)
23
+
24
+ # Set by policy_verifier
25
+ issues: List["Issue"] = field(default_factory=list)
26
+
27
+ @property
28
+ def has_issues(self) -> bool:
29
+ return bool(self.issues)
30
+
31
+
32
+ @dataclass
33
+ class Issue:
34
+ issue_type: str # 'filetype' | 'license' | 'copyright' | 'filename'
35
+ description: str
36
+ detail: str = ""
@@ -0,0 +1,136 @@
1
+ """
2
+ File type analyser — mirrors OatFileTypeAnalyser.java + Apache Rat BinaryGuesser.
3
+ Detects binary and archive files.
4
+ """
5
+ from __future__ import annotations
6
+
7
+ from pathlib import Path
8
+
9
+ # Archive file extensions (lowercase)
10
+ _ARCHIVE_EXTS = {
11
+ ".jar", ".gz", ".zip", ".tar", ".bz2", ".rar", ".war", ".7z",
12
+ ".rpm", ".deb", ".img", ".apk", ".ipa", ".whl", ".egg",
13
+ ".tar.gz", ".tar.bz2", ".tar.xz", ".tgz", ".tbz2",
14
+ }
15
+
16
+ # Binary/pre-built file extensions (lowercase).
17
+ # Mirrors OatFileUtils.PREBUILD_FILE_EXTENSION + Apache Rat BinaryGuesser extension lists.
18
+ _BINARY_EXTS = {
19
+ ".so", ".dll", ".exe", ".elf", ".bin", ".a", ".o", ".class",
20
+ ".pyc", ".pyd", ".pyo", ".hap", ".scr", ".lib", ".pdb",
21
+ ".obj", ".ko", ".d.ts", ".exp",
22
+ }
23
+
24
+ # Keystore / certificate binary extensions from Apache Rat BinaryGuesser.KEYSTORE_EXTENSIONS.
25
+ # These are always treated as binary regardless of content.
26
+ _KEYSTORE_EXTS = {
27
+ ".jks", ".keystore", ".pem", ".crl", ".truststore", ".cert", ".ks",
28
+ }
29
+
30
+ # Image extensions from Apache Rat BinaryGuesser.IMAGE_EXTENSIONS.
31
+ _IMAGE_EXTS = {
32
+ ".png", ".pdf", ".gif", ".giff", ".tif", ".tiff", ".jpg", ".jpeg",
33
+ ".ico", ".icns", ".psd", ".webp", ".bmp", ".svg",
34
+ }
35
+
36
+ # Audio / media extensions from Apache Rat BinaryGuesser.AUDIO_EXTENSIONS.
37
+ _AUDIO_EXTS = {
38
+ ".aif", ".iff", ".m3u", ".mid", ".mp3", ".mpa", ".wav", ".wma",
39
+ }
40
+
41
+ # Font / layout binary extensions
42
+ _FONT_EXTS = {
43
+ ".woff", ".woff2", ".ttf", ".eot", ".otf",
44
+ }
45
+
46
+ # Extensions that Apache Rat BinaryGuesser.NON_BINARY_EXTENSIONS explicitly lists as text.
47
+ # Files with these extensions are NEVER treated as binary by content-sniffing.
48
+ _NON_BINARY_EXTS = {
49
+ ".ac", ".am", ".bat", ".cat", ".cgi", ".classpath", ".cmd", ".config",
50
+ ".cpp", ".css", ".cwiki", ".data", ".dcl", ".dtd", ".egrm", ".ent",
51
+ ".ft", ".fn", ".fv", ".grm", ".go", ".htaccess", ".html", ".ihtml",
52
+ ".in", ".jmx", ".jsp", ".js", ".json", ".junit", ".jx",
53
+ ".manifest", ".md", ".mf", ".meta", ".mod",
54
+ ".pen", ".pl", ".pm", ".pod", ".pom", ".project", ".properties",
55
+ ".py", ".rb", ".rdf", ".rnc", ".rng", ".rnx", ".roles", ".rss",
56
+ ".sh", ".sql", ".svg", ".tld", ".txt", ".types",
57
+ ".vm", ".vsl", ".wsdd", ".wsdl", ".xargs", ".xcat", ".xconf",
58
+ ".xegrm", ".xgrm", ".xlex", ".xlog", ".xmap", ".xml",
59
+ ".xroles", ".xsamples", ".xsd", ".xsl", ".xslt", ".xsp", ".xul",
60
+ ".xweb", ".xwelcome",
61
+ # Additional common text extensions not in Apache Rat 0.13 but treated as text by OAT
62
+ ".c", ".h", ".cc", ".hh", ".java", ".ts", ".cs", ".rb", ".php",
63
+ ".yml", ".yaml", ".toml", ".rst", ".tex",
64
+ ".gradle", ".cmake", ".mk", ".makefile",
65
+ ".proto", ".thrift", ".avro",
66
+ ".ini", ".cfg", ".conf",
67
+ }
68
+
69
+ # Probe this many bytes for content-based binary detection
70
+ _PROBE_BYTES = 512
71
+
72
+ # Apache Rat HIGH_BYTES_RATIO threshold: if >30% of chars are non-ASCII → binary
73
+ _HIGH_BYTE_THRESHOLD = 30
74
+
75
+
76
+ def is_archive(path: Path) -> bool:
77
+ """Return True if the file is a known archive type."""
78
+ name = path.name.lower()
79
+ # Check compound extensions first
80
+ for ext in (".tar.gz", ".tar.bz2", ".tar.xz", ".so.gz"):
81
+ if name.endswith(ext):
82
+ return True
83
+ return path.suffix.lower() in _ARCHIVE_EXTS
84
+
85
+
86
+ def is_binary(path: Path) -> bool:
87
+ """Return True if the file should be treated as binary.
88
+
89
+ Mirrors Apache Rat BinaryGuesser.isBinary() logic used by OAT Java:
90
+ 1. Known binary extensions → binary
91
+ 2. Known keystore/image/audio/font extensions → binary
92
+ 3. Explicitly listed non-binary extensions → NOT binary (skip content check)
93
+ 4. Content check: if >30% of bytes in first 512 bytes are >127 → binary
94
+ 5. Null byte in first 512 bytes → binary
95
+ """
96
+ ext = path.suffix.lower()
97
+ name_lower = path.name.lower()
98
+
99
+ # Step 1: known binary extensions
100
+ if ext in _BINARY_EXTS:
101
+ return True
102
+
103
+ # Step 2: keystore / image / audio / font → binary
104
+ if ext in _KEYSTORE_EXTS or ext in _IMAGE_EXTS or ext in _AUDIO_EXTS or ext in _FONT_EXTS:
105
+ return True
106
+
107
+ # Step 3: explicitly non-binary extension → skip content check
108
+ if ext in _NON_BINARY_EXTS:
109
+ return False
110
+
111
+ # Step 4+5: content-based detection
112
+ try:
113
+ with open(path, "rb") as f:
114
+ chunk = f.read(_PROBE_BYTES)
115
+ if not chunk:
116
+ return False
117
+ # Null byte → binary
118
+ if b"\x00" in chunk:
119
+ return True
120
+ # High-byte ratio check (mirrors Apache Rat BinaryGuesser HIGH_BYTES_RATIO=30)
121
+ high = sum(1 for b in chunk if b > 127)
122
+ if high * 100 // len(chunk) > _HIGH_BYTE_THRESHOLD:
123
+ return True
124
+ except OSError:
125
+ pass
126
+
127
+ return False
128
+
129
+
130
+ def get_file_type(path: Path) -> str:
131
+ """Return 'archive', 'binary', or 'text'."""
132
+ if is_archive(path):
133
+ return "archive"
134
+ if is_binary(path):
135
+ return "binary"
136
+ return "text"
@@ -0,0 +1,114 @@
1
+ """
2
+ Reads the header of a text file and drives license / copyright matchers.
3
+ Mirrors OatHeaderMatchAnalyser.java + OatFileAnalyser.java.
4
+
5
+ Performance note: read_header_with_type() performs a single file open for both
6
+ binary detection and header extraction, avoiding the two-open pattern used when
7
+ calling get_file_type() + read_header() separately.
8
+ """
9
+ from __future__ import annotations
10
+
11
+ from pathlib import Path
12
+ from typing import Dict, List, Optional, Tuple
13
+
14
+ from oat.matchers.copyright_matcher import match_copyright
15
+ from oat.matchers.license_matcher import match_license
16
+
17
+ # Maximum number of header lines to read
18
+ HEADER_LINES = 100
19
+ # Maximum file size to scan (10 MB)
20
+ MAX_FILE_SIZE = 10 * 1024 * 1024
21
+
22
+ # Encodings to try, in order
23
+ _ENCODINGS = ["utf-8", "latin-1", "gbk", "utf-16"]
24
+
25
+ # Probe bytes for binary detection (mirrors Apache Rat BinaryGuesser)
26
+ _PROBE_BYTES = 512
27
+ _HIGH_BYTE_THRESHOLD = 30 # >30% non-ASCII → binary
28
+
29
+
30
+ def read_header(path: Path) -> Optional[str]:
31
+ """
32
+ Read the first HEADER_LINES lines of a text file.
33
+ Returns None if the file cannot be decoded.
34
+ """
35
+ if path.stat().st_size > MAX_FILE_SIZE:
36
+ return None
37
+
38
+ for enc in _ENCODINGS:
39
+ try:
40
+ with open(path, encoding=enc, errors="strict") as f:
41
+ lines = []
42
+ for i, line in enumerate(f):
43
+ if i >= HEADER_LINES:
44
+ break
45
+ lines.append(line)
46
+ return "".join(lines)
47
+ except (UnicodeDecodeError, LookupError):
48
+ continue
49
+ return None
50
+
51
+
52
+ def read_header_with_type(path: Path) -> Tuple[str, Optional[str]]:
53
+ """
54
+ Single-pass file read: returns (file_type, header_text_or_None).
55
+
56
+ file_type is 'text', 'binary', or 'size_exceeded'.
57
+ This avoids the double-open that occurs when get_file_type() reads 512 bytes
58
+ and then read_header() opens the file again for HEADER_LINES lines.
59
+
60
+ Callers that need the full file_type taxonomy (archive detection etc.) should
61
+ still use get_file_type() for the extension-based checks before calling this.
62
+ Only the content-sniffing part is merged here.
63
+ """
64
+ try:
65
+ size = path.stat().st_size
66
+ except OSError:
67
+ return "text", None
68
+
69
+ if size > MAX_FILE_SIZE:
70
+ return "text", None # oversized → treat as text but skip header
71
+
72
+ # --- Read raw bytes once ---
73
+ try:
74
+ with open(path, "rb") as f:
75
+ raw = f.read(MAX_FILE_SIZE)
76
+ except OSError:
77
+ return "text", None
78
+
79
+ # --- Binary sniffing (mirrors file_type.is_binary content check) ---
80
+ probe = raw[:_PROBE_BYTES]
81
+ if probe:
82
+ if b"\x00" in probe:
83
+ return "binary", None
84
+ high = sum(1 for b in probe if b > 127)
85
+ if high * 100 // len(probe) > _HIGH_BYTE_THRESHOLD:
86
+ return "binary", None
87
+
88
+ # --- Decode header lines ---
89
+ for enc in _ENCODINGS:
90
+ try:
91
+ text = raw.decode(enc, errors="strict")
92
+ lines = text.splitlines(keepends=True)[:HEADER_LINES]
93
+ return "text", "".join(lines)
94
+ except (UnicodeDecodeError, LookupError):
95
+ continue
96
+
97
+ return "text", None
98
+
99
+
100
+ def analyse_header(
101
+ path: Path,
102
+ custom_texts: Optional[Dict[str, List[str]]] = None,
103
+ ):
104
+ """
105
+ Returns (license_str, copyright_owners_list).
106
+ license_str is 'NoLicenseHeader' / 'InvalidLicense' / actual ID.
107
+ """
108
+ header = read_header(path)
109
+ if header is None:
110
+ return "NoLicenseHeader", []
111
+
112
+ license_str = match_license(header, custom_texts)
113
+ copyright_owners = match_copyright(header)
114
+ return license_str, copyright_owners
@@ -0,0 +1,122 @@
1
+ """
2
+ Analysis pipeline — mirrors OatComplianceExecutor.java + OatDefaultTaskProcessor.java.
3
+ Orchestrates: walk → file_type → header → policy_verify → collect results.
4
+ """
5
+ from __future__ import annotations
6
+
7
+ import os
8
+ from concurrent.futures import ThreadPoolExecutor, as_completed
9
+ from pathlib import Path
10
+ from typing import List
11
+
12
+ from oat.analysis.document import FileDocument
13
+ from oat.analysis.file_type import get_file_type, is_archive
14
+ from oat.analysis.header_reader import analyse_header, read_header_with_type
15
+ from oat.analysis.policy_verifier import verify
16
+ from oat.config.schema import OatConfig
17
+ from oat.reporter.model import ReportModel
18
+ from oat.walker.directory_walker import collect_files
19
+
20
+
21
+ def run_pipeline(cfg: OatConfig) -> ReportModel:
22
+ """Main entry point. Returns a populated ReportModel."""
23
+ files = collect_files(cfg)
24
+ total = len(files)
25
+ print(f"[OAT] Found {total} files to scan.")
26
+
27
+ prj = cfg.active_project
28
+ policy = prj.policy if prj else None
29
+ file_filter = prj.file_filter if prj else None
30
+ all_filters = cfg.file_filters # pass named filter dict for per-policyitem lookup
31
+
32
+ # Merge custom license texts: global + project-level
33
+ custom_texts = dict(cfg.global_license_texts)
34
+ if prj and prj.custom_license_texts:
35
+ for k, v in prj.custom_license_texts.items():
36
+ existing = custom_texts.get(k, [])
37
+ custom_texts[k] = existing + v
38
+
39
+ # Names defined by project-level licensematcher entries (e.g. "cann License").
40
+ # When the header reader identifies a file using one of these matchers, the file
41
+ # is considered to have a valid licence (mirrors Java behaviour).
42
+ custom_license_names: set = set(prj.custom_license_texts.keys()) if prj else set()
43
+
44
+ src_path = Path(cfg.src_dir)
45
+
46
+ env_val = os.getenv("OAT_MAX_WORKERS", "").strip()
47
+ if env_val.isdigit() and int(env_val) > 0:
48
+ max_workers = int(env_val)
49
+ else:
50
+ max_workers = min(32, (os.cpu_count() or 1) * 4)
51
+ print(f"[OAT] Using {max_workers} worker threads (cpu_count={os.cpu_count()}).")
52
+ docs: List[FileDocument] = []
53
+
54
+ with ThreadPoolExecutor(max_workers=max_workers) as pool:
55
+ futures = {
56
+ pool.submit(
57
+ _analyse_file,
58
+ f,
59
+ src_path,
60
+ policy,
61
+ file_filter,
62
+ all_filters,
63
+ custom_texts,
64
+ custom_license_names,
65
+ ): f
66
+ for f in files
67
+ }
68
+ done = 0
69
+ for future in as_completed(futures):
70
+ doc = future.result()
71
+ docs.append(doc)
72
+ done += 1
73
+ if done % 100 == 0 or done == total:
74
+ print(f"[OAT] Progress: {done}/{total}", end="\r", flush=True)
75
+
76
+ if total > 0:
77
+ print() # newline after progress
78
+
79
+ return ReportModel(cfg=cfg, documents=docs)
80
+
81
+
82
+ def _analyse_file(path, src_path, policy, file_filter, all_filters, custom_texts, custom_license_names) -> FileDocument:
83
+ rel = path.relative_to(src_path).as_posix()
84
+ doc = FileDocument(path=path, rel_path=rel)
85
+
86
+ # Fast path: extension-based archive detection (no I/O needed)
87
+ if is_archive(path):
88
+ doc.file_type = "archive"
89
+ else:
90
+ # Single I/O: binary sniff + header read in one pass.
91
+ # is_binary() for known extensions is pure extension lookup (no I/O);
92
+ # for unknown extensions it reads 512 bytes. read_header_with_type()
93
+ # reads the full file once and combines both steps.
94
+ from oat.analysis.file_type import _BINARY_EXTS, _KEYSTORE_EXTS, _IMAGE_EXTS, _AUDIO_EXTS, _FONT_EXTS, _NON_BINARY_EXTS # noqa: PLC0415
95
+ ext = path.suffix.lower()
96
+ if ext in _BINARY_EXTS or ext in _KEYSTORE_EXTS or ext in _IMAGE_EXTS or ext in _AUDIO_EXTS or ext in _FONT_EXTS:
97
+ # Known binary by extension — no need to read content
98
+ doc.file_type = "binary"
99
+ elif ext in _NON_BINARY_EXTS:
100
+ # Known text by extension — still need header; use plain read_header path
101
+ doc.file_type = "text"
102
+ doc.license, doc.copyright_owners = analyse_header(path, custom_texts or None)
103
+ else:
104
+ # Unknown extension: single open covers binary sniff + header extraction
105
+ ft, header_text = read_header_with_type(path)
106
+ doc.file_type = ft
107
+ if ft == "text" and header_text is not None:
108
+ from oat.matchers.license_matcher import match_license
109
+ from oat.matchers.copyright_matcher import match_copyright
110
+ doc.license = match_license(header_text, custom_texts or None)
111
+ doc.copyright_owners = match_copyright(header_text)
112
+ elif ft == "text":
113
+ doc.license = "NoLicenseHeader"
114
+
115
+ # Header analysis for known-text files (handled inline above for unknown-ext)
116
+ # This branch covers ext in _NON_BINARY_EXTS (already done) — nothing extra needed.
117
+
118
+ # Policy verification
119
+ if policy:
120
+ verify(doc, policy, file_filter, all_filters, custom_license_names)
121
+
122
+ return doc