oat-py 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- oat/__init__.py +4 -0
- oat/__main__.py +40 -0
- oat/analysis/__init__.py +0 -0
- oat/analysis/document.py +36 -0
- oat/analysis/file_type.py +136 -0
- oat/analysis/header_reader.py +114 -0
- oat/analysis/pipeline.py +122 -0
- oat/analysis/policy_verifier.py +396 -0
- oat/cli.py +100 -0
- oat/config/__init__.py +0 -0
- oat/config/loader.py +389 -0
- oat/config/schema.py +113 -0
- oat/matchers/__init__.py +0 -0
- oat/matchers/copyright_matcher.py +111 -0
- oat/matchers/license_matcher.py +472 -0
- oat/matchers/spdx_license_loader.py +67 -0
- oat/reporter/__init__.py +0 -0
- oat/reporter/model.py +33 -0
- oat/reporter/plain_reporter.py +94 -0
- oat/resources/OAT-Default.xml +138 -0
- oat/resources/__init__.py +0 -0
- oat/resources/builtin_licenses.json +532 -0
- oat/resources/licenses-exception.json +1 -0
- oat/resources/licenses.json +9804 -0
- oat/utils/__init__.py +0 -0
- oat/utils/text_util.py +37 -0
- oat/walker/__init__.py +0 -0
- oat/walker/directory_walker.py +160 -0
- oat_py-1.0.0.dist-info/METADATA +7 -0
- oat_py-1.0.0.dist-info/RECORD +33 -0
- oat_py-1.0.0.dist-info/WHEEL +5 -0
- oat_py-1.0.0.dist-info/entry_points.txt +2 -0
- oat_py-1.0.0.dist-info/top_level.txt +1 -0
oat/__init__.py
ADDED
oat/__main__.py
ADDED
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Entry point: python -m oat or oat (if installed via pip)
|
|
3
|
+
"""
|
|
4
|
+
import sys
|
|
5
|
+
import time
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def main(argv=None):
|
|
10
|
+
from oat.cli import parse_args
|
|
11
|
+
args = parse_args(argv)
|
|
12
|
+
|
|
13
|
+
start = time.time()
|
|
14
|
+
print(f"[OAT] OpenHarmony OSS Audit Tool - Python Edition")
|
|
15
|
+
print(f"[OAT] Scanning: {args.s}")
|
|
16
|
+
|
|
17
|
+
from oat.config.loader import build_task
|
|
18
|
+
from oat.analysis.pipeline import run_pipeline
|
|
19
|
+
from oat.reporter.plain_reporter import PlainReporter
|
|
20
|
+
|
|
21
|
+
task = build_task(args)
|
|
22
|
+
|
|
23
|
+
report_model = run_pipeline(task)
|
|
24
|
+
|
|
25
|
+
reporter = PlainReporter(task, report_model)
|
|
26
|
+
reporter.write()
|
|
27
|
+
|
|
28
|
+
elapsed = time.time() - start
|
|
29
|
+
total = report_model.total_issues
|
|
30
|
+
print(f"[OAT] Scan complete in {elapsed:.1f}s. Issues found: {total}")
|
|
31
|
+
if total > 0:
|
|
32
|
+
print(f"[OAT] Report: {task.report_dir}")
|
|
33
|
+
sys.exit(1)
|
|
34
|
+
else:
|
|
35
|
+
print("[OAT] No compliance issues found.")
|
|
36
|
+
sys.exit(0)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
if __name__ == "__main__":
|
|
40
|
+
main()
|
oat/analysis/__init__.py
ADDED
|
File without changes
|
oat/analysis/document.py
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
"""
|
|
2
|
+
FileDocument — mirrors OatFileDocument.java.
|
|
3
|
+
Holds analysis results for a single file.
|
|
4
|
+
"""
|
|
5
|
+
from __future__ import annotations
|
|
6
|
+
|
|
7
|
+
from dataclasses import dataclass, field
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
from typing import List, Optional
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
@dataclass
|
|
13
|
+
class FileDocument:
|
|
14
|
+
path: Path
|
|
15
|
+
rel_path: str = "" # relative to src_dir, forward slashes
|
|
16
|
+
|
|
17
|
+
# Set by file_type analyser
|
|
18
|
+
file_type: str = "text" # 'text' | 'binary' | 'archive'
|
|
19
|
+
|
|
20
|
+
# Set by header analyser
|
|
21
|
+
license: str = "NoLicenseHeader"
|
|
22
|
+
copyright_owners: List[str] = field(default_factory=list)
|
|
23
|
+
|
|
24
|
+
# Set by policy_verifier
|
|
25
|
+
issues: List["Issue"] = field(default_factory=list)
|
|
26
|
+
|
|
27
|
+
@property
|
|
28
|
+
def has_issues(self) -> bool:
|
|
29
|
+
return bool(self.issues)
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
@dataclass
|
|
33
|
+
class Issue:
|
|
34
|
+
issue_type: str # 'filetype' | 'license' | 'copyright' | 'filename'
|
|
35
|
+
description: str
|
|
36
|
+
detail: str = ""
|
|
@@ -0,0 +1,136 @@
|
|
|
1
|
+
"""
|
|
2
|
+
File type analyser — mirrors OatFileTypeAnalyser.java + Apache Rat BinaryGuesser.
|
|
3
|
+
Detects binary and archive files.
|
|
4
|
+
"""
|
|
5
|
+
from __future__ import annotations
|
|
6
|
+
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
# Archive file extensions (lowercase)
|
|
10
|
+
_ARCHIVE_EXTS = {
|
|
11
|
+
".jar", ".gz", ".zip", ".tar", ".bz2", ".rar", ".war", ".7z",
|
|
12
|
+
".rpm", ".deb", ".img", ".apk", ".ipa", ".whl", ".egg",
|
|
13
|
+
".tar.gz", ".tar.bz2", ".tar.xz", ".tgz", ".tbz2",
|
|
14
|
+
}
|
|
15
|
+
|
|
16
|
+
# Binary/pre-built file extensions (lowercase).
|
|
17
|
+
# Mirrors OatFileUtils.PREBUILD_FILE_EXTENSION + Apache Rat BinaryGuesser extension lists.
|
|
18
|
+
_BINARY_EXTS = {
|
|
19
|
+
".so", ".dll", ".exe", ".elf", ".bin", ".a", ".o", ".class",
|
|
20
|
+
".pyc", ".pyd", ".pyo", ".hap", ".scr", ".lib", ".pdb",
|
|
21
|
+
".obj", ".ko", ".d.ts", ".exp",
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
# Keystore / certificate binary extensions from Apache Rat BinaryGuesser.KEYSTORE_EXTENSIONS.
|
|
25
|
+
# These are always treated as binary regardless of content.
|
|
26
|
+
_KEYSTORE_EXTS = {
|
|
27
|
+
".jks", ".keystore", ".pem", ".crl", ".truststore", ".cert", ".ks",
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
# Image extensions from Apache Rat BinaryGuesser.IMAGE_EXTENSIONS.
|
|
31
|
+
_IMAGE_EXTS = {
|
|
32
|
+
".png", ".pdf", ".gif", ".giff", ".tif", ".tiff", ".jpg", ".jpeg",
|
|
33
|
+
".ico", ".icns", ".psd", ".webp", ".bmp", ".svg",
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
# Audio / media extensions from Apache Rat BinaryGuesser.AUDIO_EXTENSIONS.
|
|
37
|
+
_AUDIO_EXTS = {
|
|
38
|
+
".aif", ".iff", ".m3u", ".mid", ".mp3", ".mpa", ".wav", ".wma",
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
# Font / layout binary extensions
|
|
42
|
+
_FONT_EXTS = {
|
|
43
|
+
".woff", ".woff2", ".ttf", ".eot", ".otf",
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
# Extensions that Apache Rat BinaryGuesser.NON_BINARY_EXTENSIONS explicitly lists as text.
|
|
47
|
+
# Files with these extensions are NEVER treated as binary by content-sniffing.
|
|
48
|
+
_NON_BINARY_EXTS = {
|
|
49
|
+
".ac", ".am", ".bat", ".cat", ".cgi", ".classpath", ".cmd", ".config",
|
|
50
|
+
".cpp", ".css", ".cwiki", ".data", ".dcl", ".dtd", ".egrm", ".ent",
|
|
51
|
+
".ft", ".fn", ".fv", ".grm", ".go", ".htaccess", ".html", ".ihtml",
|
|
52
|
+
".in", ".jmx", ".jsp", ".js", ".json", ".junit", ".jx",
|
|
53
|
+
".manifest", ".md", ".mf", ".meta", ".mod",
|
|
54
|
+
".pen", ".pl", ".pm", ".pod", ".pom", ".project", ".properties",
|
|
55
|
+
".py", ".rb", ".rdf", ".rnc", ".rng", ".rnx", ".roles", ".rss",
|
|
56
|
+
".sh", ".sql", ".svg", ".tld", ".txt", ".types",
|
|
57
|
+
".vm", ".vsl", ".wsdd", ".wsdl", ".xargs", ".xcat", ".xconf",
|
|
58
|
+
".xegrm", ".xgrm", ".xlex", ".xlog", ".xmap", ".xml",
|
|
59
|
+
".xroles", ".xsamples", ".xsd", ".xsl", ".xslt", ".xsp", ".xul",
|
|
60
|
+
".xweb", ".xwelcome",
|
|
61
|
+
# Additional common text extensions not in Apache Rat 0.13 but treated as text by OAT
|
|
62
|
+
".c", ".h", ".cc", ".hh", ".java", ".ts", ".cs", ".rb", ".php",
|
|
63
|
+
".yml", ".yaml", ".toml", ".rst", ".tex",
|
|
64
|
+
".gradle", ".cmake", ".mk", ".makefile",
|
|
65
|
+
".proto", ".thrift", ".avro",
|
|
66
|
+
".ini", ".cfg", ".conf",
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
# Probe this many bytes for content-based binary detection
|
|
70
|
+
_PROBE_BYTES = 512
|
|
71
|
+
|
|
72
|
+
# Apache Rat HIGH_BYTES_RATIO threshold: if >30% of chars are non-ASCII → binary
|
|
73
|
+
_HIGH_BYTE_THRESHOLD = 30
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def is_archive(path: Path) -> bool:
|
|
77
|
+
"""Return True if the file is a known archive type."""
|
|
78
|
+
name = path.name.lower()
|
|
79
|
+
# Check compound extensions first
|
|
80
|
+
for ext in (".tar.gz", ".tar.bz2", ".tar.xz", ".so.gz"):
|
|
81
|
+
if name.endswith(ext):
|
|
82
|
+
return True
|
|
83
|
+
return path.suffix.lower() in _ARCHIVE_EXTS
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def is_binary(path: Path) -> bool:
|
|
87
|
+
"""Return True if the file should be treated as binary.
|
|
88
|
+
|
|
89
|
+
Mirrors Apache Rat BinaryGuesser.isBinary() logic used by OAT Java:
|
|
90
|
+
1. Known binary extensions → binary
|
|
91
|
+
2. Known keystore/image/audio/font extensions → binary
|
|
92
|
+
3. Explicitly listed non-binary extensions → NOT binary (skip content check)
|
|
93
|
+
4. Content check: if >30% of bytes in first 512 bytes are >127 → binary
|
|
94
|
+
5. Null byte in first 512 bytes → binary
|
|
95
|
+
"""
|
|
96
|
+
ext = path.suffix.lower()
|
|
97
|
+
name_lower = path.name.lower()
|
|
98
|
+
|
|
99
|
+
# Step 1: known binary extensions
|
|
100
|
+
if ext in _BINARY_EXTS:
|
|
101
|
+
return True
|
|
102
|
+
|
|
103
|
+
# Step 2: keystore / image / audio / font → binary
|
|
104
|
+
if ext in _KEYSTORE_EXTS or ext in _IMAGE_EXTS or ext in _AUDIO_EXTS or ext in _FONT_EXTS:
|
|
105
|
+
return True
|
|
106
|
+
|
|
107
|
+
# Step 3: explicitly non-binary extension → skip content check
|
|
108
|
+
if ext in _NON_BINARY_EXTS:
|
|
109
|
+
return False
|
|
110
|
+
|
|
111
|
+
# Step 4+5: content-based detection
|
|
112
|
+
try:
|
|
113
|
+
with open(path, "rb") as f:
|
|
114
|
+
chunk = f.read(_PROBE_BYTES)
|
|
115
|
+
if not chunk:
|
|
116
|
+
return False
|
|
117
|
+
# Null byte → binary
|
|
118
|
+
if b"\x00" in chunk:
|
|
119
|
+
return True
|
|
120
|
+
# High-byte ratio check (mirrors Apache Rat BinaryGuesser HIGH_BYTES_RATIO=30)
|
|
121
|
+
high = sum(1 for b in chunk if b > 127)
|
|
122
|
+
if high * 100 // len(chunk) > _HIGH_BYTE_THRESHOLD:
|
|
123
|
+
return True
|
|
124
|
+
except OSError:
|
|
125
|
+
pass
|
|
126
|
+
|
|
127
|
+
return False
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def get_file_type(path: Path) -> str:
|
|
131
|
+
"""Return 'archive', 'binary', or 'text'."""
|
|
132
|
+
if is_archive(path):
|
|
133
|
+
return "archive"
|
|
134
|
+
if is_binary(path):
|
|
135
|
+
return "binary"
|
|
136
|
+
return "text"
|
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Reads the header of a text file and drives license / copyright matchers.
|
|
3
|
+
Mirrors OatHeaderMatchAnalyser.java + OatFileAnalyser.java.
|
|
4
|
+
|
|
5
|
+
Performance note: read_header_with_type() performs a single file open for both
|
|
6
|
+
binary detection and header extraction, avoiding the two-open pattern used when
|
|
7
|
+
calling get_file_type() + read_header() separately.
|
|
8
|
+
"""
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
from typing import Dict, List, Optional, Tuple
|
|
13
|
+
|
|
14
|
+
from oat.matchers.copyright_matcher import match_copyright
|
|
15
|
+
from oat.matchers.license_matcher import match_license
|
|
16
|
+
|
|
17
|
+
# Maximum number of header lines to read
|
|
18
|
+
HEADER_LINES = 100
|
|
19
|
+
# Maximum file size to scan (10 MB)
|
|
20
|
+
MAX_FILE_SIZE = 10 * 1024 * 1024
|
|
21
|
+
|
|
22
|
+
# Encodings to try, in order
|
|
23
|
+
_ENCODINGS = ["utf-8", "latin-1", "gbk", "utf-16"]
|
|
24
|
+
|
|
25
|
+
# Probe bytes for binary detection (mirrors Apache Rat BinaryGuesser)
|
|
26
|
+
_PROBE_BYTES = 512
|
|
27
|
+
_HIGH_BYTE_THRESHOLD = 30 # >30% non-ASCII → binary
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def read_header(path: Path) -> Optional[str]:
|
|
31
|
+
"""
|
|
32
|
+
Read the first HEADER_LINES lines of a text file.
|
|
33
|
+
Returns None if the file cannot be decoded.
|
|
34
|
+
"""
|
|
35
|
+
if path.stat().st_size > MAX_FILE_SIZE:
|
|
36
|
+
return None
|
|
37
|
+
|
|
38
|
+
for enc in _ENCODINGS:
|
|
39
|
+
try:
|
|
40
|
+
with open(path, encoding=enc, errors="strict") as f:
|
|
41
|
+
lines = []
|
|
42
|
+
for i, line in enumerate(f):
|
|
43
|
+
if i >= HEADER_LINES:
|
|
44
|
+
break
|
|
45
|
+
lines.append(line)
|
|
46
|
+
return "".join(lines)
|
|
47
|
+
except (UnicodeDecodeError, LookupError):
|
|
48
|
+
continue
|
|
49
|
+
return None
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def read_header_with_type(path: Path) -> Tuple[str, Optional[str]]:
|
|
53
|
+
"""
|
|
54
|
+
Single-pass file read: returns (file_type, header_text_or_None).
|
|
55
|
+
|
|
56
|
+
file_type is 'text', 'binary', or 'size_exceeded'.
|
|
57
|
+
This avoids the double-open that occurs when get_file_type() reads 512 bytes
|
|
58
|
+
and then read_header() opens the file again for HEADER_LINES lines.
|
|
59
|
+
|
|
60
|
+
Callers that need the full file_type taxonomy (archive detection etc.) should
|
|
61
|
+
still use get_file_type() for the extension-based checks before calling this.
|
|
62
|
+
Only the content-sniffing part is merged here.
|
|
63
|
+
"""
|
|
64
|
+
try:
|
|
65
|
+
size = path.stat().st_size
|
|
66
|
+
except OSError:
|
|
67
|
+
return "text", None
|
|
68
|
+
|
|
69
|
+
if size > MAX_FILE_SIZE:
|
|
70
|
+
return "text", None # oversized → treat as text but skip header
|
|
71
|
+
|
|
72
|
+
# --- Read raw bytes once ---
|
|
73
|
+
try:
|
|
74
|
+
with open(path, "rb") as f:
|
|
75
|
+
raw = f.read(MAX_FILE_SIZE)
|
|
76
|
+
except OSError:
|
|
77
|
+
return "text", None
|
|
78
|
+
|
|
79
|
+
# --- Binary sniffing (mirrors file_type.is_binary content check) ---
|
|
80
|
+
probe = raw[:_PROBE_BYTES]
|
|
81
|
+
if probe:
|
|
82
|
+
if b"\x00" in probe:
|
|
83
|
+
return "binary", None
|
|
84
|
+
high = sum(1 for b in probe if b > 127)
|
|
85
|
+
if high * 100 // len(probe) > _HIGH_BYTE_THRESHOLD:
|
|
86
|
+
return "binary", None
|
|
87
|
+
|
|
88
|
+
# --- Decode header lines ---
|
|
89
|
+
for enc in _ENCODINGS:
|
|
90
|
+
try:
|
|
91
|
+
text = raw.decode(enc, errors="strict")
|
|
92
|
+
lines = text.splitlines(keepends=True)[:HEADER_LINES]
|
|
93
|
+
return "text", "".join(lines)
|
|
94
|
+
except (UnicodeDecodeError, LookupError):
|
|
95
|
+
continue
|
|
96
|
+
|
|
97
|
+
return "text", None
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def analyse_header(
|
|
101
|
+
path: Path,
|
|
102
|
+
custom_texts: Optional[Dict[str, List[str]]] = None,
|
|
103
|
+
):
|
|
104
|
+
"""
|
|
105
|
+
Returns (license_str, copyright_owners_list).
|
|
106
|
+
license_str is 'NoLicenseHeader' / 'InvalidLicense' / actual ID.
|
|
107
|
+
"""
|
|
108
|
+
header = read_header(path)
|
|
109
|
+
if header is None:
|
|
110
|
+
return "NoLicenseHeader", []
|
|
111
|
+
|
|
112
|
+
license_str = match_license(header, custom_texts)
|
|
113
|
+
copyright_owners = match_copyright(header)
|
|
114
|
+
return license_str, copyright_owners
|
oat/analysis/pipeline.py
ADDED
|
@@ -0,0 +1,122 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Analysis pipeline — mirrors OatComplianceExecutor.java + OatDefaultTaskProcessor.java.
|
|
3
|
+
Orchestrates: walk → file_type → header → policy_verify → collect results.
|
|
4
|
+
"""
|
|
5
|
+
from __future__ import annotations
|
|
6
|
+
|
|
7
|
+
import os
|
|
8
|
+
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from typing import List
|
|
11
|
+
|
|
12
|
+
from oat.analysis.document import FileDocument
|
|
13
|
+
from oat.analysis.file_type import get_file_type, is_archive
|
|
14
|
+
from oat.analysis.header_reader import analyse_header, read_header_with_type
|
|
15
|
+
from oat.analysis.policy_verifier import verify
|
|
16
|
+
from oat.config.schema import OatConfig
|
|
17
|
+
from oat.reporter.model import ReportModel
|
|
18
|
+
from oat.walker.directory_walker import collect_files
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def run_pipeline(cfg: OatConfig) -> ReportModel:
|
|
22
|
+
"""Main entry point. Returns a populated ReportModel."""
|
|
23
|
+
files = collect_files(cfg)
|
|
24
|
+
total = len(files)
|
|
25
|
+
print(f"[OAT] Found {total} files to scan.")
|
|
26
|
+
|
|
27
|
+
prj = cfg.active_project
|
|
28
|
+
policy = prj.policy if prj else None
|
|
29
|
+
file_filter = prj.file_filter if prj else None
|
|
30
|
+
all_filters = cfg.file_filters # pass named filter dict for per-policyitem lookup
|
|
31
|
+
|
|
32
|
+
# Merge custom license texts: global + project-level
|
|
33
|
+
custom_texts = dict(cfg.global_license_texts)
|
|
34
|
+
if prj and prj.custom_license_texts:
|
|
35
|
+
for k, v in prj.custom_license_texts.items():
|
|
36
|
+
existing = custom_texts.get(k, [])
|
|
37
|
+
custom_texts[k] = existing + v
|
|
38
|
+
|
|
39
|
+
# Names defined by project-level licensematcher entries (e.g. "cann License").
|
|
40
|
+
# When the header reader identifies a file using one of these matchers, the file
|
|
41
|
+
# is considered to have a valid licence (mirrors Java behaviour).
|
|
42
|
+
custom_license_names: set = set(prj.custom_license_texts.keys()) if prj else set()
|
|
43
|
+
|
|
44
|
+
src_path = Path(cfg.src_dir)
|
|
45
|
+
|
|
46
|
+
env_val = os.getenv("OAT_MAX_WORKERS", "").strip()
|
|
47
|
+
if env_val.isdigit() and int(env_val) > 0:
|
|
48
|
+
max_workers = int(env_val)
|
|
49
|
+
else:
|
|
50
|
+
max_workers = min(32, (os.cpu_count() or 1) * 4)
|
|
51
|
+
print(f"[OAT] Using {max_workers} worker threads (cpu_count={os.cpu_count()}).")
|
|
52
|
+
docs: List[FileDocument] = []
|
|
53
|
+
|
|
54
|
+
with ThreadPoolExecutor(max_workers=max_workers) as pool:
|
|
55
|
+
futures = {
|
|
56
|
+
pool.submit(
|
|
57
|
+
_analyse_file,
|
|
58
|
+
f,
|
|
59
|
+
src_path,
|
|
60
|
+
policy,
|
|
61
|
+
file_filter,
|
|
62
|
+
all_filters,
|
|
63
|
+
custom_texts,
|
|
64
|
+
custom_license_names,
|
|
65
|
+
): f
|
|
66
|
+
for f in files
|
|
67
|
+
}
|
|
68
|
+
done = 0
|
|
69
|
+
for future in as_completed(futures):
|
|
70
|
+
doc = future.result()
|
|
71
|
+
docs.append(doc)
|
|
72
|
+
done += 1
|
|
73
|
+
if done % 100 == 0 or done == total:
|
|
74
|
+
print(f"[OAT] Progress: {done}/{total}", end="\r", flush=True)
|
|
75
|
+
|
|
76
|
+
if total > 0:
|
|
77
|
+
print() # newline after progress
|
|
78
|
+
|
|
79
|
+
return ReportModel(cfg=cfg, documents=docs)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def _analyse_file(path, src_path, policy, file_filter, all_filters, custom_texts, custom_license_names) -> FileDocument:
|
|
83
|
+
rel = path.relative_to(src_path).as_posix()
|
|
84
|
+
doc = FileDocument(path=path, rel_path=rel)
|
|
85
|
+
|
|
86
|
+
# Fast path: extension-based archive detection (no I/O needed)
|
|
87
|
+
if is_archive(path):
|
|
88
|
+
doc.file_type = "archive"
|
|
89
|
+
else:
|
|
90
|
+
# Single I/O: binary sniff + header read in one pass.
|
|
91
|
+
# is_binary() for known extensions is pure extension lookup (no I/O);
|
|
92
|
+
# for unknown extensions it reads 512 bytes. read_header_with_type()
|
|
93
|
+
# reads the full file once and combines both steps.
|
|
94
|
+
from oat.analysis.file_type import _BINARY_EXTS, _KEYSTORE_EXTS, _IMAGE_EXTS, _AUDIO_EXTS, _FONT_EXTS, _NON_BINARY_EXTS # noqa: PLC0415
|
|
95
|
+
ext = path.suffix.lower()
|
|
96
|
+
if ext in _BINARY_EXTS or ext in _KEYSTORE_EXTS or ext in _IMAGE_EXTS or ext in _AUDIO_EXTS or ext in _FONT_EXTS:
|
|
97
|
+
# Known binary by extension — no need to read content
|
|
98
|
+
doc.file_type = "binary"
|
|
99
|
+
elif ext in _NON_BINARY_EXTS:
|
|
100
|
+
# Known text by extension — still need header; use plain read_header path
|
|
101
|
+
doc.file_type = "text"
|
|
102
|
+
doc.license, doc.copyright_owners = analyse_header(path, custom_texts or None)
|
|
103
|
+
else:
|
|
104
|
+
# Unknown extension: single open covers binary sniff + header extraction
|
|
105
|
+
ft, header_text = read_header_with_type(path)
|
|
106
|
+
doc.file_type = ft
|
|
107
|
+
if ft == "text" and header_text is not None:
|
|
108
|
+
from oat.matchers.license_matcher import match_license
|
|
109
|
+
from oat.matchers.copyright_matcher import match_copyright
|
|
110
|
+
doc.license = match_license(header_text, custom_texts or None)
|
|
111
|
+
doc.copyright_owners = match_copyright(header_text)
|
|
112
|
+
elif ft == "text":
|
|
113
|
+
doc.license = "NoLicenseHeader"
|
|
114
|
+
|
|
115
|
+
# Header analysis for known-text files (handled inline above for unknown-ext)
|
|
116
|
+
# This branch covers ext in _NON_BINARY_EXTS (already done) — nothing extra needed.
|
|
117
|
+
|
|
118
|
+
# Policy verification
|
|
119
|
+
if policy:
|
|
120
|
+
verify(doc, policy, file_filter, all_filters, custom_license_names)
|
|
121
|
+
|
|
122
|
+
return doc
|