fosslight-source 2.3.9__tar.gz → 2.3.11__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {fosslight_source-2.3.9/src/fosslight_source.egg-info → fosslight_source-2.3.11}/PKG-INFO +3 -2
- {fosslight_source-2.3.9 → fosslight_source-2.3.11}/pyproject.toml +3 -2
- fosslight_source-2.3.11/src/fosslight_source/_exclude.py +21 -0
- {fosslight_source-2.3.9 → fosslight_source-2.3.11}/src/fosslight_source/_help.py +2 -0
- {fosslight_source-2.3.9 → fosslight_source-2.3.11}/src/fosslight_source/_kb_client.py +28 -1
- {fosslight_source-2.3.9 → fosslight_source-2.3.11}/src/fosslight_source/_merge.py +2 -1
- {fosslight_source-2.3.9 → fosslight_source-2.3.11}/src/fosslight_source/_parsing_scancode_file_item.py +69 -30
- {fosslight_source-2.3.9 → fosslight_source-2.3.11}/src/fosslight_source/_parsing_scanoss_file.py +2 -1
- {fosslight_source-2.3.9 → fosslight_source-2.3.11}/src/fosslight_source/_scan_item.py +6 -4
- {fosslight_source-2.3.9 → fosslight_source-2.3.11}/src/fosslight_source/_scancode_ignore_binaries.py +43 -8
- {fosslight_source-2.3.9 → fosslight_source-2.3.11}/src/fosslight_source/cli.py +9 -5
- {fosslight_source-2.3.9 → fosslight_source-2.3.11}/src/fosslight_source/run_scancode.py +42 -1
- {fosslight_source-2.3.9 → fosslight_source-2.3.11/src/fosslight_source.egg-info}/PKG-INFO +3 -2
- {fosslight_source-2.3.9 → fosslight_source-2.3.11}/src/fosslight_source.egg-info/SOURCES.txt +3 -0
- {fosslight_source-2.3.9 → fosslight_source-2.3.11}/src/fosslight_source.egg-info/requires.txt +2 -1
- fosslight_source-2.3.11/tests/test_kb_ssl.py +52 -0
- {fosslight_source-2.3.9 → fosslight_source-2.3.11}/tests/test_manifest_android_bp.py +1 -1
- {fosslight_source-2.3.9 → fosslight_source-2.3.11}/tests/test_manifest_recommended_scenarios.py +1 -1
- fosslight_source-2.3.11/tests/test_multi_value_order.py +26 -0
- {fosslight_source-2.3.9 → fosslight_source-2.3.11}/tests/test_parsing_unknown_spdx.py +172 -8
- {fosslight_source-2.3.9 → fosslight_source-2.3.11}/LICENSE +0 -0
- {fosslight_source-2.3.9 → fosslight_source-2.3.11}/MANIFEST.in +0 -0
- {fosslight_source-2.3.9 → fosslight_source-2.3.11}/README.md +0 -0
- {fosslight_source-2.3.9 → fosslight_source-2.3.11}/setup.cfg +0 -0
- {fosslight_source-2.3.9 → fosslight_source-2.3.11}/src/fosslight_source/__init__.py +0 -0
- {fosslight_source-2.3.9 → fosslight_source-2.3.11}/src/fosslight_source/_license_matched.py +0 -0
- {fosslight_source-2.3.9 → fosslight_source-2.3.11}/src/fosslight_source/run_manifest_extractor.py +0 -0
- {fosslight_source-2.3.9 → fosslight_source-2.3.11}/src/fosslight_source/run_scanoss.py +0 -0
- {fosslight_source-2.3.9 → fosslight_source-2.3.11}/src/fosslight_source/run_spdx_extractor.py +0 -0
- {fosslight_source-2.3.9 → fosslight_source-2.3.11}/src/fosslight_source.egg-info/dependency_links.txt +0 -0
- {fosslight_source-2.3.9 → fosslight_source-2.3.11}/src/fosslight_source.egg-info/entry_points.txt +0 -0
- {fosslight_source-2.3.9 → fosslight_source-2.3.11}/src/fosslight_source.egg-info/top_level.txt +0 -0
- {fosslight_source-2.3.9 → fosslight_source-2.3.11}/tests/test_manifest_composer.py +0 -0
- {fosslight_source-2.3.9 → fosslight_source-2.3.11}/tests/test_manifest_pyproject.py +0 -0
- {fosslight_source-2.3.9 → fosslight_source-2.3.11}/tests/test_tox.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: fosslight_source
|
|
3
|
-
Version: 2.3.
|
|
3
|
+
Version: 2.3.11
|
|
4
4
|
Summary: FOSSLight Source Scanner
|
|
5
5
|
Author: LG Electronics
|
|
6
6
|
License-Expression: Apache-2.0
|
|
@@ -20,7 +20,7 @@ Requires-Dist: setuptools<=80.10.2
|
|
|
20
20
|
Requires-Dist: pyparsing
|
|
21
21
|
Requires-Dist: scanoss>=1.45.0
|
|
22
22
|
Requires-Dist: XlsxWriter
|
|
23
|
-
Requires-Dist: fosslight_util>=2.2.
|
|
23
|
+
Requires-Dist: fosslight_util>=2.2.12
|
|
24
24
|
Requires-Dist: PyYAML
|
|
25
25
|
Requires-Dist: wheel>=0.38.1
|
|
26
26
|
Requires-Dist: intbitset
|
|
@@ -32,6 +32,7 @@ Requires-Dist: normality==2.6.1
|
|
|
32
32
|
Requires-Dist: psycopg2-binary>=2.9.10; python_version >= "3.13"
|
|
33
33
|
Requires-Dist: tomli; python_version < "3.11"
|
|
34
34
|
Requires-Dist: tqdm
|
|
35
|
+
Requires-Dist: truststore
|
|
35
36
|
Dynamic: license-file
|
|
36
37
|
|
|
37
38
|
<!--
|
|
@@ -7,7 +7,7 @@ build-backend = "setuptools.build_meta"
|
|
|
7
7
|
|
|
8
8
|
[project]
|
|
9
9
|
name = "fosslight_source"
|
|
10
|
-
version = "2.3.
|
|
10
|
+
version = "2.3.11"
|
|
11
11
|
description = "FOSSLight Source Scanner"
|
|
12
12
|
readme = "README.md"
|
|
13
13
|
license = "Apache-2.0"
|
|
@@ -29,7 +29,7 @@ dependencies = [
|
|
|
29
29
|
"pyparsing",
|
|
30
30
|
"scanoss>=1.45.0",
|
|
31
31
|
"XlsxWriter",
|
|
32
|
-
"fosslight_util>=2.2.
|
|
32
|
+
"fosslight_util>=2.2.12",
|
|
33
33
|
"PyYAML",
|
|
34
34
|
"wheel>=0.38.1",
|
|
35
35
|
"intbitset",
|
|
@@ -43,6 +43,7 @@ dependencies = [
|
|
|
43
43
|
"psycopg2-binary>=2.9.10; python_version >= '3.13'",
|
|
44
44
|
"tomli; python_version < '3.11'",
|
|
45
45
|
"tqdm",
|
|
46
|
+
"truststore",
|
|
46
47
|
]
|
|
47
48
|
|
|
48
49
|
[project.optional-dependencies]
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
# Copyright (c) 2026 LG Electronics Inc.
|
|
2
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
3
|
+
|
|
4
|
+
"""Source-scanner filename excludes (not applied to dependency discovery)."""
|
|
5
|
+
|
|
6
|
+
from fosslight_util.exclude import is_excluded_filename
|
|
7
|
+
|
|
8
|
+
# Build/package config noise for license scans.
|
|
9
|
+
EXCLUDE_FILENAME_SOURCE = frozenset({
|
|
10
|
+
"changelog", "config.guess", "config.sub", "changes", "ltmain.sh",
|
|
11
|
+
"configure", "configure.ac", "depcomp", "compile", "missing", "makefile",
|
|
12
|
+
"makefile.am",
|
|
13
|
+
"build.gradle", "build.gradle.kts", "settings.gradle", "settings.gradle.kts",
|
|
14
|
+
"gradlew", "gradlew.bat",
|
|
15
|
+
"vite.config.ts", "vite.config.js", "vite.config.mts", "vite.config.mjs",
|
|
16
|
+
"package-lock.json", "npm-shrinkwrap.json", "yarn.lock", "pnpm-lock.yaml",
|
|
17
|
+
})
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def is_excluded_source_filename(file_path: str) -> bool:
|
|
21
|
+
return is_excluded_filename(file_path, EXCLUDE_FILENAME_SOURCE)
|
|
@@ -45,6 +45,8 @@ _HELP_MESSAGE_SOURCE_SCANNER = f"""
|
|
|
45
45
|
--hide_progress Hide the progress bar during scanning
|
|
46
46
|
--kb_url <url> KB API URL (priority: parameter > KB_URL env > default)
|
|
47
47
|
--kb_token <token> KB bearer token (priority: parameter > KB_TOKEN env)
|
|
48
|
+
HTTPS uses the OS certificate store. Set KB_SSL_VERIFY=false
|
|
49
|
+
only if certificate verification must be skipped.
|
|
48
50
|
|
|
49
51
|
💡 Examples
|
|
50
52
|
────────────────────────────────────────────────────────────────────
|
|
@@ -5,6 +5,8 @@
|
|
|
5
5
|
|
|
6
6
|
import json
|
|
7
7
|
import logging
|
|
8
|
+
import os
|
|
9
|
+
import ssl
|
|
8
10
|
import time
|
|
9
11
|
import urllib.error
|
|
10
12
|
import urllib.request
|
|
@@ -19,6 +21,31 @@ _SCAN_JOB_POLL_MAX_INTERVAL_SEC = 10.0
|
|
|
19
21
|
_SCAN_JOB_REQUEST_TIMEOUT_SEC = 30
|
|
20
22
|
_SCAN_JOB_MIN_WAIT_SEC = 300
|
|
21
23
|
_SCAN_JOB_PER_HASH_SEC = 35
|
|
24
|
+
_KB_SSL_VERIFY_FALSE = {"0", "false", "no", "off"}
|
|
25
|
+
_logged_insecure_kb_ssl = False
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def kb_ssl_verify_enabled() -> bool:
|
|
29
|
+
raw = os.environ.get("KB_SSL_VERIFY", "true")
|
|
30
|
+
return raw.strip().lower() not in _KB_SSL_VERIFY_FALSE
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def create_kb_ssl_context() -> ssl.SSLContext:
|
|
34
|
+
"""TLS context for KB HTTPS. Uses the OS trust store (Windows store, like the browser)."""
|
|
35
|
+
global _logged_insecure_kb_ssl
|
|
36
|
+
if not kb_ssl_verify_enabled():
|
|
37
|
+
if not _logged_insecure_kb_ssl:
|
|
38
|
+
logger.warning("KB TLS certificate verification is disabled (KB_SSL_VERIFY)")
|
|
39
|
+
_logged_insecure_kb_ssl = True
|
|
40
|
+
ctx = ssl.SSLContext(ssl.PROTOCOL_TLS_CLIENT)
|
|
41
|
+
ctx.check_hostname = False
|
|
42
|
+
ctx.verify_mode = ssl.CERT_NONE
|
|
43
|
+
return ctx
|
|
44
|
+
try:
|
|
45
|
+
import truststore
|
|
46
|
+
return truststore.SSLContext(ssl.PROTOCOL_TLS_CLIENT)
|
|
47
|
+
except ImportError:
|
|
48
|
+
return ssl.create_default_context()
|
|
22
49
|
|
|
23
50
|
|
|
24
51
|
def _kb_request(
|
|
@@ -40,7 +67,7 @@ def _kb_request(
|
|
|
40
67
|
if kb_token:
|
|
41
68
|
request.add_header("Authorization", f"Bearer {kb_token}")
|
|
42
69
|
|
|
43
|
-
with urllib.request.urlopen(request, timeout=timeout) as response:
|
|
70
|
+
with urllib.request.urlopen(request, timeout=timeout, context=create_kb_ssl_context()) as response:
|
|
44
71
|
body = response.read().decode()
|
|
45
72
|
return json.loads(body) if body else {}
|
|
46
73
|
|
|
@@ -111,7 +111,8 @@ def _get_top_merge_values(scan_items: list, value_getter) -> list:
|
|
|
111
111
|
normalized_value = _normalize_merge_text(value)
|
|
112
112
|
if normalized_value:
|
|
113
113
|
values.append(normalized_value)
|
|
114
|
-
|
|
114
|
+
ranked = sorted(Counter(values).items(), key=lambda entry: (-entry[1], entry[0]))
|
|
115
|
+
return [value for value, _ in ranked[:3]]
|
|
115
116
|
|
|
116
117
|
|
|
117
118
|
def _can_merge_folder(scan_items: list) -> bool:
|
|
@@ -6,8 +6,9 @@
|
|
|
6
6
|
import os
|
|
7
7
|
import logging
|
|
8
8
|
import re
|
|
9
|
+
from functools import lru_cache
|
|
9
10
|
import fosslight_util.constant as constant
|
|
10
|
-
from
|
|
11
|
+
from ._exclude import is_excluded_source_filename
|
|
11
12
|
from ._license_matched import MatchedLicense
|
|
12
13
|
from ._scan_item import SourceItem
|
|
13
14
|
from ._scan_item import replace_word
|
|
@@ -34,6 +35,7 @@ KEYWORD_SPDX_ID = r'SPDX-License-Identifier\s*[\S]+'
|
|
|
34
35
|
KEYWORD_DOWNLOAD_LOC = r'DownloadLocation\s*[\S]+'
|
|
35
36
|
KEYWORD_SCANCODE_UNKNOWN = "unknown-spdx"
|
|
36
37
|
KEYWORD_UNKNOWN_LICENSE_REFERENCE = "unknown-license-reference"
|
|
38
|
+
HTTP_URL_PATTERN = re.compile(r'https?://', re.IGNORECASE)
|
|
37
39
|
SPDX_REPLACE_WORDS = ["(", ")"]
|
|
38
40
|
KEY_AND_OR = re.compile(r"(?<=\s)(?:and|or)(?=\s)", re.IGNORECASE)
|
|
39
41
|
KEY_AND_OR_CAPTURE = re.compile(r"(?<=\s)(and|or)(?=\s)", re.IGNORECASE)
|
|
@@ -61,25 +63,41 @@ def filter_fsf_copyright_from_gpl_license_text(
|
|
|
61
63
|
return [c for c in copyrights if FSF_IN_COPYRIGHT not in c.lower()]
|
|
62
64
|
|
|
63
65
|
|
|
64
|
-
def
|
|
66
|
+
def _expression_has_other_license(license_expression: str) -> bool:
|
|
65
67
|
if not license_expression:
|
|
66
68
|
return False
|
|
67
69
|
for token in split_spdx_expression(license_expression.lower()):
|
|
68
70
|
token = token.strip()
|
|
69
|
-
if
|
|
71
|
+
if (
|
|
72
|
+
token
|
|
73
|
+
and KEYWORD_UNKNOWN_LICENSE_REFERENCE not in token
|
|
74
|
+
and token not in REMOVE_LICENSE
|
|
75
|
+
):
|
|
70
76
|
return True
|
|
71
77
|
return False
|
|
72
78
|
|
|
73
79
|
|
|
74
|
-
def
|
|
75
|
-
"""
|
|
76
|
-
texts = set()
|
|
80
|
+
def _file_has_other_license(matches: list) -> bool:
|
|
81
|
+
"""Return whether a file has a license other than unknown-license-reference."""
|
|
77
82
|
for matched_lic in matches or []:
|
|
78
|
-
matched_txt = matched_lic.get("matched_text") or ""
|
|
79
83
|
license_expression = matched_lic.get("license_expression") or ""
|
|
80
|
-
if
|
|
81
|
-
|
|
82
|
-
return
|
|
84
|
+
if _expression_has_other_license(license_expression):
|
|
85
|
+
return True
|
|
86
|
+
return False
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def _matched_text_has_http_url(matched_text: str) -> bool:
|
|
90
|
+
"""Return whether matched text contains an HTTP or HTTPS URL."""
|
|
91
|
+
return bool(HTTP_URL_PATTERN.search(matched_text or ""))
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _should_keep_unknown_license_reference(
|
|
95
|
+
matched_text: str, has_other_license_in_file: bool
|
|
96
|
+
) -> bool:
|
|
97
|
+
return (
|
|
98
|
+
not has_other_license_in_file
|
|
99
|
+
and _matched_text_has_http_url(matched_text)
|
|
100
|
+
)
|
|
83
101
|
|
|
84
102
|
|
|
85
103
|
def get_error_from_header(header_item: list) -> Tuple[bool, str]:
|
|
@@ -424,23 +442,25 @@ def _build_unknown_spdx_replacement_queue(matches: list) -> list[str]:
|
|
|
424
442
|
|
|
425
443
|
|
|
426
444
|
def _should_suppress_unknown_license_reference(
|
|
427
|
-
matches: list,
|
|
445
|
+
matches: list, has_other_license_in_file: bool
|
|
428
446
|
) -> bool:
|
|
447
|
+
has_unknown_license_reference = False
|
|
429
448
|
for matched_lic in matches or []:
|
|
430
449
|
expr = (matched_lic.get("license_expression") or "").lower()
|
|
431
|
-
|
|
432
|
-
|
|
433
|
-
|
|
434
|
-
|
|
450
|
+
if KEYWORD_UNKNOWN_LICENSE_REFERENCE not in expr:
|
|
451
|
+
continue
|
|
452
|
+
has_unknown_license_reference = True
|
|
453
|
+
if _should_keep_unknown_license_reference(
|
|
454
|
+
matched_lic.get("matched_text") or "", has_other_license_in_file
|
|
435
455
|
):
|
|
436
|
-
return
|
|
437
|
-
return
|
|
456
|
+
return False
|
|
457
|
+
return has_unknown_license_reference
|
|
438
458
|
|
|
439
459
|
|
|
440
460
|
def build_comment_from_detected_expression(
|
|
441
461
|
detected_expression: str,
|
|
442
462
|
matches: list,
|
|
443
|
-
|
|
463
|
+
suppress_unknown_license_reference: bool,
|
|
444
464
|
) -> str:
|
|
445
465
|
"""
|
|
446
466
|
Rebuild comment from detected expression.
|
|
@@ -470,26 +490,35 @@ def build_comment_from_detected_expression(
|
|
|
470
490
|
)
|
|
471
491
|
|
|
472
492
|
replacements = _build_unknown_spdx_replacement_queue(matches)
|
|
473
|
-
suppress_ulr = _should_suppress_unknown_license_reference(
|
|
474
|
-
matches, matched_texts_with_other_licenses
|
|
475
|
-
)
|
|
476
493
|
tree = _parse_license_expression_tokens(_tokenize_license_expression(expr))
|
|
477
494
|
if tree is None:
|
|
478
495
|
return ""
|
|
479
496
|
transformed = _transform_license_expr_node(
|
|
480
|
-
tree, replacements, [0],
|
|
497
|
+
tree, replacements, [0], suppress_unknown_license_reference
|
|
481
498
|
)
|
|
482
499
|
return _omit_parens_for_two_license_expression(
|
|
483
500
|
_serialize_license_expr_node(transformed)
|
|
484
501
|
)
|
|
485
502
|
|
|
486
503
|
|
|
504
|
+
@lru_cache(maxsize=65536)
|
|
487
505
|
def get_license_expression_spdx(license_expression: str) -> str:
|
|
488
506
|
if not license_expression or not license_expression.strip():
|
|
489
507
|
return ""
|
|
490
508
|
try:
|
|
491
|
-
from licensedcode.cache import
|
|
492
|
-
|
|
509
|
+
from licensedcode.cache import (
|
|
510
|
+
build_spdx_license_expression,
|
|
511
|
+
get_licenses_db,
|
|
512
|
+
get_licensing,
|
|
513
|
+
)
|
|
514
|
+
expression = license_expression.strip()
|
|
515
|
+
# licensedcode re-reads the whole license database from disk for every
|
|
516
|
+
# key it cannot resolve, and then raises anyway. Screen the keys first
|
|
517
|
+
# so unknown tokens stay cheap.
|
|
518
|
+
licenses_db = get_licenses_db()
|
|
519
|
+
if any(key not in licenses_db for key in get_licensing().license_keys(expression)):
|
|
520
|
+
return ""
|
|
521
|
+
result = build_spdx_license_expression(expression)
|
|
493
522
|
if result is None:
|
|
494
523
|
return ""
|
|
495
524
|
if isinstance(result, str) and result.lower().startswith("licenseref-"):
|
|
@@ -516,7 +545,7 @@ def parsing_scancode(
|
|
|
516
545
|
if (not file_path) or is_binary or is_dir:
|
|
517
546
|
logger.info(f"Skipping {file_path} because it is binary or directory")
|
|
518
547
|
continue
|
|
519
|
-
if
|
|
548
|
+
if is_excluded_source_filename(file_path):
|
|
520
549
|
logger.debug(f"Skipping {file_path} because it is an excluded filename")
|
|
521
550
|
continue
|
|
522
551
|
result_item = SourceItem(file_path)
|
|
@@ -545,7 +574,12 @@ def parsing_scancode(
|
|
|
545
574
|
all_matches = []
|
|
546
575
|
for lic in licenses or []:
|
|
547
576
|
all_matches.extend(lic.get("matches") or [])
|
|
548
|
-
|
|
577
|
+
has_other_license_in_file = _file_has_other_license(all_matches)
|
|
578
|
+
suppress_unknown_license_reference = (
|
|
579
|
+
_should_suppress_unknown_license_reference(
|
|
580
|
+
all_matches, has_other_license_in_file
|
|
581
|
+
)
|
|
582
|
+
)
|
|
549
583
|
for lic in licenses or []:
|
|
550
584
|
matched_lic_list = lic.get("matches", [])
|
|
551
585
|
for matched_lic in matched_lic_list:
|
|
@@ -565,7 +599,9 @@ def parsing_scancode(
|
|
|
565
599
|
continue
|
|
566
600
|
if (
|
|
567
601
|
KEYWORD_UNKNOWN_LICENSE_REFERENCE in found_lic.lower()
|
|
568
|
-
and
|
|
602
|
+
and not _should_keep_unknown_license_reference(
|
|
603
|
+
matched_txt, has_other_license_in_file
|
|
604
|
+
)
|
|
569
605
|
):
|
|
570
606
|
continue
|
|
571
607
|
if KEYWORD_SCANCODE_UNKNOWN in found_lic.lower():
|
|
@@ -592,8 +628,10 @@ def parsing_scancode(
|
|
|
592
628
|
file.get("percentage_of_license_text", 0) > 90 and not is_source_file
|
|
593
629
|
)
|
|
594
630
|
|
|
595
|
-
result_item.copyright =
|
|
596
|
-
|
|
631
|
+
result_item.copyright = sorted(
|
|
632
|
+
filter_fsf_copyright_from_gpl_license_text(
|
|
633
|
+
copyright_value_list, license_detected, result_item.is_license_text
|
|
634
|
+
)
|
|
597
635
|
)
|
|
598
636
|
|
|
599
637
|
if len(license_detected) > 1:
|
|
@@ -603,6 +641,7 @@ def parsing_scancode(
|
|
|
603
641
|
resolved_unknown_spdx
|
|
604
642
|
or KEYWORD_SCANCODE_UNKNOWN in detected_expression.lower()
|
|
605
643
|
or "licenseref-scancode-unknown-spdx" in detected_expression_spdx.lower()
|
|
644
|
+
or suppress_unknown_license_reference
|
|
606
645
|
):
|
|
607
646
|
# Prefer non-SPDX expression so unknown-spdx tokens map cleanly.
|
|
608
647
|
# Comment only for dual-license style expressions that include OR.
|
|
@@ -611,7 +650,7 @@ def parsing_scancode(
|
|
|
611
650
|
result_item.comment = build_comment_from_detected_expression(
|
|
612
651
|
source_expression,
|
|
613
652
|
all_matches,
|
|
614
|
-
|
|
653
|
+
suppress_unknown_license_reference,
|
|
615
654
|
)
|
|
616
655
|
else:
|
|
617
656
|
license_expression = detected_expression_spdx or detected_expression
|
{fosslight_source-2.3.9 → fosslight_source-2.3.11}/src/fosslight_source/_parsing_scanoss_file.py
RENAMED
|
@@ -5,6 +5,7 @@
|
|
|
5
5
|
|
|
6
6
|
import logging
|
|
7
7
|
import fosslight_util.constant as constant
|
|
8
|
+
from ._exclude import is_excluded_source_filename
|
|
8
9
|
from ._scan_item import SourceItem
|
|
9
10
|
from ._scan_item import replace_word
|
|
10
11
|
from typing import Tuple
|
|
@@ -39,7 +40,7 @@ def parsing_scan_result(scanoss_report: dict, excluded_files: set = None) -> Tup
|
|
|
39
40
|
|
|
40
41
|
for file_path, findings in scanoss_report.items():
|
|
41
42
|
file_path_normalized = file_path.replace('\\', '/')
|
|
42
|
-
if file_path_normalized in excluded_files:
|
|
43
|
+
if file_path_normalized in excluded_files or is_excluded_source_filename(file_path_normalized):
|
|
43
44
|
continue
|
|
44
45
|
result_item = SourceItem(file_path)
|
|
45
46
|
|
|
@@ -7,6 +7,7 @@ import os
|
|
|
7
7
|
import logging
|
|
8
8
|
import re
|
|
9
9
|
import hashlib
|
|
10
|
+
from bisect import insort
|
|
10
11
|
import fosslight_util.constant as constant
|
|
11
12
|
from fosslight_util.oss_item import FileItem, OssItem, get_checksum_sha1
|
|
12
13
|
|
|
@@ -78,13 +79,13 @@ class SourceItem(FileItem):
|
|
|
78
79
|
def licenses(self, value: list) -> None:
|
|
79
80
|
if value:
|
|
80
81
|
max_length_exceed = False
|
|
81
|
-
for new_lic in value:
|
|
82
|
+
for new_lic in sorted(value):
|
|
82
83
|
if new_lic:
|
|
83
84
|
if len(new_lic) > MAX_LICENSE_LENGTH:
|
|
84
85
|
new_lic = new_lic[:MAX_LICENSE_LENGTH]
|
|
85
86
|
max_length_exceed = True
|
|
86
87
|
if new_lic not in self._licenses:
|
|
87
|
-
self._licenses
|
|
88
|
+
insort(self._licenses, new_lic)
|
|
88
89
|
if len(",".join(self._licenses)) > MAX_LICENSE_TOTAL_LENGTH:
|
|
89
90
|
self._licenses.remove(new_lic)
|
|
90
91
|
max_length_exceed = True
|
|
@@ -180,10 +181,11 @@ class SourceItem(FileItem):
|
|
|
180
181
|
kb_origin_urls: dict[str, str] | None = None,
|
|
181
182
|
) -> None:
|
|
182
183
|
self.oss_items = []
|
|
184
|
+
copyrights = "\n".join(self.copyright)
|
|
183
185
|
if self.download_location:
|
|
184
186
|
for url in self.download_location:
|
|
185
187
|
item = OssItem(self.oss_name, self.oss_version, self.licenses, url)
|
|
186
|
-
item.copyright =
|
|
188
|
+
item.copyright = copyrights
|
|
187
189
|
item.comment = self.comment
|
|
188
190
|
self.oss_items.append(item)
|
|
189
191
|
else:
|
|
@@ -198,7 +200,7 @@ class SourceItem(FileItem):
|
|
|
198
200
|
oss_name, oss_version, download_url = self._apply_kb_origin_url(origin_url)
|
|
199
201
|
item = OssItem(oss_name, oss_version, self.licenses, download_url)
|
|
200
202
|
|
|
201
|
-
item.copyright =
|
|
203
|
+
item.copyright = copyrights
|
|
202
204
|
item.comment = self.comment
|
|
203
205
|
self.oss_items.append(item)
|
|
204
206
|
|
{fosslight_source-2.3.9 → fosslight_source-2.3.11}/src/fosslight_source/_scancode_ignore_binaries.py
RENAMED
|
@@ -6,12 +6,22 @@
|
|
|
6
6
|
# so PyPI installs do not need a GitHub git dependency.
|
|
7
7
|
# SPDX-PackageDownloadLocation: https://github.com/aboutcode-org/scancode-plugins/tree/main/misc/scancode-ignore-binaries
|
|
8
8
|
|
|
9
|
+
import logging
|
|
10
|
+
import multiprocessing
|
|
11
|
+
|
|
9
12
|
from plugincode.pre_scan import PreScanPlugin
|
|
10
13
|
from plugincode.pre_scan import pre_scan_impl
|
|
11
14
|
from commoncode.cliutils import PluggableCommandLineOption
|
|
12
15
|
from commoncode.cliutils import PRE_SCAN_GROUP
|
|
13
16
|
from typecode.contenttype import get_type
|
|
14
17
|
|
|
18
|
+
logger = logging.getLogger(__name__)
|
|
19
|
+
|
|
20
|
+
# Detection costs one get_type() call per file, which reads from disk. On large
|
|
21
|
+
# trees that dominates the pre-scan stage, so spread it over the scan processes.
|
|
22
|
+
PARALLEL_MIN_FILES = 2000
|
|
23
|
+
CHUNK_SIZE = 256
|
|
24
|
+
|
|
15
25
|
|
|
16
26
|
@pre_scan_impl
|
|
17
27
|
class IgnoreBinaries(PreScanPlugin):
|
|
@@ -32,22 +42,47 @@ class IgnoreBinaries(PreScanPlugin):
|
|
|
32
42
|
def is_enabled(self, ignore_binaries, **kwargs):
|
|
33
43
|
return ignore_binaries
|
|
34
44
|
|
|
35
|
-
def process_codebase(self, codebase, ignore_binaries, **kwargs):
|
|
45
|
+
def process_codebase(self, codebase, ignore_binaries, processes=1, **kwargs):
|
|
36
46
|
"""
|
|
37
47
|
Remove binary Resources from the resource tree.
|
|
38
48
|
"""
|
|
39
49
|
if not ignore_binaries:
|
|
40
50
|
return
|
|
41
51
|
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
52
|
+
# Collect first: removing while walking would invalidate the walk.
|
|
53
|
+
candidates = [
|
|
54
|
+
(resource.path, resource.location)
|
|
55
|
+
for resource in codebase.walk()
|
|
56
|
+
if resource.is_file
|
|
57
|
+
]
|
|
58
|
+
if not candidates:
|
|
59
|
+
return
|
|
60
|
+
|
|
61
|
+
locations = [location for _path, location in candidates]
|
|
62
|
+
flags = _detect_binaries(locations, processes)
|
|
63
|
+
|
|
64
|
+
for (path, _location), binary in zip(candidates, flags):
|
|
65
|
+
if not binary:
|
|
45
66
|
continue
|
|
46
|
-
|
|
47
|
-
|
|
67
|
+
resource = codebase.get_resource(path)
|
|
68
|
+
if resource is not None:
|
|
69
|
+
resource.remove(codebase)
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _detect_binaries(locations, processes):
|
|
73
|
+
"""
|
|
74
|
+
Return a list of booleans, one per location, telling whether it is binary.
|
|
75
|
+
"""
|
|
76
|
+
if processes and processes > 1 and len(locations) >= PARALLEL_MIN_FILES:
|
|
77
|
+
try:
|
|
78
|
+
pool = multiprocessing.Pool(processes)
|
|
79
|
+
except Exception as ex:
|
|
80
|
+
logger.debug(f"Parallel binary detection unavailable, using one process: {ex}")
|
|
81
|
+
else:
|
|
82
|
+
with pool:
|
|
83
|
+
return pool.map(is_binary, locations, chunksize=CHUNK_SIZE)
|
|
48
84
|
|
|
49
|
-
|
|
50
|
-
resource.remove(codebase)
|
|
85
|
+
return [is_binary(location) for location in locations]
|
|
51
86
|
|
|
52
87
|
|
|
53
88
|
def is_binary(location):
|
|
@@ -22,6 +22,7 @@ from fosslight_util.correct import correct_with_yaml
|
|
|
22
22
|
from fosslight_util.parsing_yaml import SUPPORT_OSS_INFO_FILES
|
|
23
23
|
from .run_scancode import run_scan
|
|
24
24
|
from fosslight_util.exclude import get_excluded_paths
|
|
25
|
+
from ._exclude import EXCLUDE_FILENAME_SOURCE, is_excluded_source_filename
|
|
25
26
|
from .run_scanoss import run_scanoss_py
|
|
26
27
|
from .run_scanoss import get_scanoss_extra_info
|
|
27
28
|
import yaml
|
|
@@ -30,7 +31,7 @@ import argparse
|
|
|
30
31
|
from .run_spdx_extractor import get_spdx_downloads
|
|
31
32
|
from .run_manifest_extractor import get_manifest_licenses
|
|
32
33
|
from ._scan_item import SourceItem, resolve_kb_config, is_notice_file, is_manifest_file
|
|
33
|
-
from ._kb_client import fetch_origin_urls_via_scan_job
|
|
34
|
+
from ._kb_client import create_kb_ssl_context, fetch_origin_urls_via_scan_job
|
|
34
35
|
from fosslight_util.cover import dump_result_log
|
|
35
36
|
from fosslight_util.time import current_timestamp_utc, format_running_time, timestamp_for_filename
|
|
36
37
|
from fosslight_util.oss_item import ScannerItem
|
|
@@ -295,7 +296,7 @@ def check_kb_server_reachable(kb_url: str, kb_token: str = "") -> bool:
|
|
|
295
296
|
request = urllib.request.Request(f"{kb_url}health", method='GET')
|
|
296
297
|
if kb_token:
|
|
297
298
|
request.add_header('Authorization', f'Bearer {kb_token}')
|
|
298
|
-
with urllib.request.urlopen(request, timeout=10) as response:
|
|
299
|
+
with urllib.request.urlopen(request, timeout=10, context=create_kb_ssl_context()) as response:
|
|
299
300
|
logger.debug(f"KB server is reachable. Response status: {response.status}")
|
|
300
301
|
return True
|
|
301
302
|
except urllib.error.HTTPError:
|
|
@@ -380,7 +381,8 @@ def _collect_kb_file_hashes(
|
|
|
380
381
|
|
|
381
382
|
for file_path in tqdm.tqdm(files_to_scan, desc="KB Hashing", disable=hide_progress):
|
|
382
383
|
rel_path = os.path.relpath(file_path, abs_path_to_scan).replace("\\", "/")
|
|
383
|
-
if rel_path in scancode_paths or rel_path in excluded_files
|
|
384
|
+
if (rel_path in scancode_paths or rel_path in excluded_files
|
|
385
|
+
or is_excluded_source_filename(rel_path) or is_notice_file(file_path)):
|
|
384
386
|
continue
|
|
385
387
|
extra_item = SourceItem(rel_path)
|
|
386
388
|
md5_hash, _wfp = extra_item._get_hash(path_to_scan)
|
|
@@ -629,7 +631,9 @@ def run_scanners(
|
|
|
629
631
|
(excluded_path_with_default_exclusion,
|
|
630
632
|
excluded_path_without_dot,
|
|
631
633
|
excluded_files,
|
|
632
|
-
cnt_file_except_skipped) = get_excluded_paths(
|
|
634
|
+
cnt_file_except_skipped) = get_excluded_paths(
|
|
635
|
+
path_to_scan, path_to_exclude_with_filename,
|
|
636
|
+
exclude_filenames=EXCLUDE_FILENAME_SOURCE)
|
|
633
637
|
logger.debug(f"Skipped paths count: {len(excluded_path_with_default_exclusion)}")
|
|
634
638
|
|
|
635
639
|
if not selected_scanner:
|
|
@@ -727,7 +731,7 @@ def metadata_collector(path_to_scan: str, excluded_files: set) -> tuple[dict, di
|
|
|
727
731
|
for file in files:
|
|
728
732
|
file_path = os.path.join(root, file)
|
|
729
733
|
rel_path_file = os.path.relpath(file_path, abs_path_to_scan).replace('\\', '/')
|
|
730
|
-
if rel_path_file in excluded_files:
|
|
734
|
+
if rel_path_file in excluded_files or is_excluded_source_filename(rel_path_file):
|
|
731
735
|
continue
|
|
732
736
|
|
|
733
737
|
downloads = get_spdx_downloads(file_path)
|
|
@@ -26,6 +26,11 @@ logger = logging.getLogger(constant.LOGGER_NAME)
|
|
|
26
26
|
warnings.filterwarnings("ignore", category=FutureWarning)
|
|
27
27
|
_PKG_NAME = "fosslight_source"
|
|
28
28
|
|
|
29
|
+
_MAX_IN_MEMORY_ENV = "FOSSLIGHT_SCANCODE_MAX_IN_MEMORY"
|
|
30
|
+
_SCANCODE_DEFAULT_MAX_IN_MEMORY = 10000
|
|
31
|
+
_MEMORY_FRACTION_FOR_CODEBASE = 0.4
|
|
32
|
+
_BYTES_PER_RESOURCE = 4096
|
|
33
|
+
|
|
29
34
|
try:
|
|
30
35
|
from click.core import UNSET as _CLICK_UNSET # Click >= 8.3
|
|
31
36
|
_HAS_CLICK_UNSET = True
|
|
@@ -165,7 +170,7 @@ def _default_scancode_ignore_patterns(
|
|
|
165
170
|
Directory names use path-based globs (e.g. **/tests/**) so they do not match
|
|
166
171
|
the scan root directory name itself.
|
|
167
172
|
Binary files are excluded separately via scancode --ignore-binaries.
|
|
168
|
-
|
|
173
|
+
EXCLUDE_FILENAME_SOURCE is not passed to ScanCode --ignore: matching many exact
|
|
169
174
|
names on a large tree is slow. Those files are dropped after parsing.
|
|
170
175
|
"""
|
|
171
176
|
patterns = {".*"}
|
|
@@ -182,6 +187,39 @@ def _default_scancode_ignore_patterns(
|
|
|
182
187
|
return tuple(sorted(patterns))
|
|
183
188
|
|
|
184
189
|
|
|
190
|
+
def _available_memory_bytes() -> int:
|
|
191
|
+
try:
|
|
192
|
+
with open("/proc/meminfo") as meminfo:
|
|
193
|
+
for line in meminfo:
|
|
194
|
+
if line.startswith("MemAvailable:"):
|
|
195
|
+
return int(line.split()[1]) * 1024
|
|
196
|
+
except OSError:
|
|
197
|
+
pass
|
|
198
|
+
try:
|
|
199
|
+
return os.sysconf("SC_PAGE_SIZE") * os.sysconf("SC_AVPHYS_PAGES")
|
|
200
|
+
except (AttributeError, ValueError, OSError):
|
|
201
|
+
return 0
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def _resolve_max_in_memory() -> int:
|
|
205
|
+
"""
|
|
206
|
+
Past --max-in-memory resources, ScanCode caches every Resource in its own
|
|
207
|
+
file under the temp dir, which makes the inventory and every codebase walk
|
|
208
|
+
disk bound. Raise the threshold to whatever memory allows and let ScanCode
|
|
209
|
+
keep spilling beyond that.
|
|
210
|
+
"""
|
|
211
|
+
override = os.environ.get(_MAX_IN_MEMORY_ENV, "").strip()
|
|
212
|
+
if override:
|
|
213
|
+
try:
|
|
214
|
+
return int(override)
|
|
215
|
+
except ValueError:
|
|
216
|
+
logger.warning(f"Ignoring invalid {_MAX_IN_MEMORY_ENV}={override}")
|
|
217
|
+
|
|
218
|
+
budget = _available_memory_bytes() * _MEMORY_FRACTION_FOR_CODEBASE
|
|
219
|
+
resolved = int(budget // _BYTES_PER_RESOURCE)
|
|
220
|
+
return max(resolved, _SCANCODE_DEFAULT_MAX_IN_MEMORY)
|
|
221
|
+
|
|
222
|
+
|
|
185
223
|
def run_scan(
|
|
186
224
|
path_to_scan: str, output_file_name: str = "",
|
|
187
225
|
_write_json_file: bool = False, num_cores: int = -1,
|
|
@@ -252,9 +290,12 @@ def run_scan(
|
|
|
252
290
|
path_to_exclude, abs_path_to_scan
|
|
253
291
|
)
|
|
254
292
|
logger.debug(f"Scancode ignore patterns: {len(ignore_tuple)}")
|
|
293
|
+
max_in_memory = _resolve_max_in_memory()
|
|
294
|
+
logger.debug(f"Scancode max_in_memory: {max_in_memory}")
|
|
255
295
|
|
|
256
296
|
kwargs = {
|
|
257
297
|
"max_depth": 100,
|
|
298
|
+
"max_in_memory": max_in_memory,
|
|
258
299
|
"strip_root": True,
|
|
259
300
|
"license": True,
|
|
260
301
|
"copyright": True,
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: fosslight_source
|
|
3
|
-
Version: 2.3.
|
|
3
|
+
Version: 2.3.11
|
|
4
4
|
Summary: FOSSLight Source Scanner
|
|
5
5
|
Author: LG Electronics
|
|
6
6
|
License-Expression: Apache-2.0
|
|
@@ -20,7 +20,7 @@ Requires-Dist: setuptools<=80.10.2
|
|
|
20
20
|
Requires-Dist: pyparsing
|
|
21
21
|
Requires-Dist: scanoss>=1.45.0
|
|
22
22
|
Requires-Dist: XlsxWriter
|
|
23
|
-
Requires-Dist: fosslight_util>=2.2.
|
|
23
|
+
Requires-Dist: fosslight_util>=2.2.12
|
|
24
24
|
Requires-Dist: PyYAML
|
|
25
25
|
Requires-Dist: wheel>=0.38.1
|
|
26
26
|
Requires-Dist: intbitset
|
|
@@ -32,6 +32,7 @@ Requires-Dist: normality==2.6.1
|
|
|
32
32
|
Requires-Dist: psycopg2-binary>=2.9.10; python_version >= "3.13"
|
|
33
33
|
Requires-Dist: tomli; python_version < "3.11"
|
|
34
34
|
Requires-Dist: tqdm
|
|
35
|
+
Requires-Dist: truststore
|
|
35
36
|
Dynamic: license-file
|
|
36
37
|
|
|
37
38
|
<!--
|
{fosslight_source-2.3.9 → fosslight_source-2.3.11}/src/fosslight_source.egg-info/SOURCES.txt
RENAMED
|
@@ -3,6 +3,7 @@ MANIFEST.in
|
|
|
3
3
|
README.md
|
|
4
4
|
pyproject.toml
|
|
5
5
|
src/fosslight_source/__init__.py
|
|
6
|
+
src/fosslight_source/_exclude.py
|
|
6
7
|
src/fosslight_source/_help.py
|
|
7
8
|
src/fosslight_source/_kb_client.py
|
|
8
9
|
src/fosslight_source/_license_matched.py
|
|
@@ -22,9 +23,11 @@ src/fosslight_source.egg-info/dependency_links.txt
|
|
|
22
23
|
src/fosslight_source.egg-info/entry_points.txt
|
|
23
24
|
src/fosslight_source.egg-info/requires.txt
|
|
24
25
|
src/fosslight_source.egg-info/top_level.txt
|
|
26
|
+
tests/test_kb_ssl.py
|
|
25
27
|
tests/test_manifest_android_bp.py
|
|
26
28
|
tests/test_manifest_composer.py
|
|
27
29
|
tests/test_manifest_pyproject.py
|
|
28
30
|
tests/test_manifest_recommended_scenarios.py
|
|
31
|
+
tests/test_multi_value_order.py
|
|
29
32
|
tests/test_parsing_unknown_spdx.py
|
|
30
33
|
tests/test_tox.py
|
{fosslight_source-2.3.9 → fosslight_source-2.3.11}/src/fosslight_source.egg-info/requires.txt
RENAMED
|
@@ -2,7 +2,7 @@ setuptools<=80.10.2
|
|
|
2
2
|
pyparsing
|
|
3
3
|
scanoss>=1.45.0
|
|
4
4
|
XlsxWriter
|
|
5
|
-
fosslight_util>=2.2.
|
|
5
|
+
fosslight_util>=2.2.12
|
|
6
6
|
PyYAML
|
|
7
7
|
wheel>=0.38.1
|
|
8
8
|
intbitset
|
|
@@ -11,6 +11,7 @@ lxml>=6.0.1
|
|
|
11
11
|
fingerprints==1.2.3
|
|
12
12
|
normality==2.6.1
|
|
13
13
|
tqdm
|
|
14
|
+
truststore
|
|
14
15
|
|
|
15
16
|
[:platform_system == "Darwin" and platform_machine == "x86_64"]
|
|
16
17
|
cryptography<49
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
# Copyright (c) 2026 LG Electronics Inc.
|
|
2
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
3
|
+
"""Tests for KB HTTPS SSL context selection."""
|
|
4
|
+
|
|
5
|
+
import ssl
|
|
6
|
+
from unittest.mock import patch
|
|
7
|
+
|
|
8
|
+
from fosslight_source._kb_client import create_kb_ssl_context, kb_ssl_verify_enabled
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def test_kb_ssl_verify_enabled_default(monkeypatch):
|
|
12
|
+
monkeypatch.delenv("KB_SSL_VERIFY", raising=False)
|
|
13
|
+
assert kb_ssl_verify_enabled() is True
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def test_kb_ssl_verify_disabled_values(monkeypatch):
|
|
17
|
+
for value in ("false", "0", "no", "OFF"):
|
|
18
|
+
monkeypatch.setenv("KB_SSL_VERIFY", value)
|
|
19
|
+
assert kb_ssl_verify_enabled() is False
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def test_create_kb_ssl_context_insecure(monkeypatch):
|
|
23
|
+
monkeypatch.setenv("KB_SSL_VERIFY", "false")
|
|
24
|
+
ctx = create_kb_ssl_context()
|
|
25
|
+
assert ctx.verify_mode == ssl.CERT_NONE
|
|
26
|
+
assert ctx.check_hostname is False
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def test_create_kb_ssl_context_uses_truststore_when_verify_on(monkeypatch):
|
|
30
|
+
monkeypatch.setenv("KB_SSL_VERIFY", "true")
|
|
31
|
+
ctx = create_kb_ssl_context()
|
|
32
|
+
assert ctx.verify_mode != ssl.CERT_NONE
|
|
33
|
+
assert ctx.check_hostname is True
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def test_kb_request_passes_ssl_context():
|
|
37
|
+
from fosslight_source._kb_client import _kb_request
|
|
38
|
+
|
|
39
|
+
class _Resp:
|
|
40
|
+
def read(self):
|
|
41
|
+
return b"{}"
|
|
42
|
+
|
|
43
|
+
def __enter__(self):
|
|
44
|
+
return self
|
|
45
|
+
|
|
46
|
+
def __exit__(self, *args):
|
|
47
|
+
return False
|
|
48
|
+
|
|
49
|
+
with patch("fosslight_source._kb_client.urllib.request.urlopen", return_value=_Resp()) as urlopen:
|
|
50
|
+
_kb_request("https://kb.example/", "health")
|
|
51
|
+
assert urlopen.call_args.kwargs["context"] is not None
|
|
52
|
+
assert isinstance(urlopen.call_args.kwargs["context"], ssl.SSLContext)
|
|
@@ -49,7 +49,7 @@ def test_merge_results_sets_manifest_flag_without_overwriting_scancode_licenses(
|
|
|
49
49
|
|
|
50
50
|
assert len(merged) == 1
|
|
51
51
|
assert merged[0].is_manifest_file is True
|
|
52
|
-
assert merged[0].licenses == ["Apache-2.0", "
|
|
52
|
+
assert merged[0].licenses == ["Apache-2.0", "BSD", "MIT"]
|
|
53
53
|
|
|
54
54
|
|
|
55
55
|
def test_merge_results_skips_android_bp_not_in_scancode_result():
|
{fosslight_source-2.3.9 → fosslight_source-2.3.11}/tests/test_manifest_recommended_scenarios.py
RENAMED
|
@@ -18,7 +18,7 @@ def test_scenario1_android_bp_keeps_scancode_licenses():
|
|
|
18
18
|
|
|
19
19
|
assert len(merged) == 1
|
|
20
20
|
assert merged[0].is_manifest_file is True
|
|
21
|
-
assert merged[0].licenses == ["Apache-2.0", "
|
|
21
|
+
assert merged[0].licenses == ["Apache-2.0", "BSD", "MIT", "OFL", "unknown-license-reference"]
|
|
22
22
|
|
|
23
23
|
|
|
24
24
|
def test_scenario2_package_json_manifest_fail_keeps_scancode_licenses():
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
# Copyright (c) 2026 LG Electronics Inc.
|
|
2
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
3
|
+
"""Stable order for multi-value License / Copyright cells."""
|
|
4
|
+
|
|
5
|
+
from fosslight_source._merge import _get_top_merge_values
|
|
6
|
+
from fosslight_source._scan_item import SourceItem
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def test_licenses_are_stored_sorted():
|
|
10
|
+
item = SourceItem("dummy.c")
|
|
11
|
+
item.licenses = ["zlib", "mit", "apache-2.0"]
|
|
12
|
+
assert item.licenses == ["apache-2.0", "mit", "zlib"]
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def test_top_merge_copyrights_break_count_ties_alphabetically():
|
|
16
|
+
items = []
|
|
17
|
+
for text in ["Copyright Z", "Copyright A", "Copyright Z", "Copyright M"]:
|
|
18
|
+
item = SourceItem("dummy.c")
|
|
19
|
+
item.copyright = [text]
|
|
20
|
+
items.append(item)
|
|
21
|
+
|
|
22
|
+
assert _get_top_merge_values(items, lambda i: i.copyright) == [
|
|
23
|
+
"Copyright Z",
|
|
24
|
+
"Copyright A",
|
|
25
|
+
"Copyright M",
|
|
26
|
+
]
|
|
@@ -9,7 +9,9 @@ from fosslight_source._parsing_scancode_file_item import (
|
|
|
9
9
|
_declared_licenses_from_matched_text,
|
|
10
10
|
build_comment_from_detected_expression,
|
|
11
11
|
parsing_scancode,
|
|
12
|
-
|
|
12
|
+
_file_has_other_license,
|
|
13
|
+
_matched_text_has_http_url,
|
|
14
|
+
_should_suppress_unknown_license_reference,
|
|
13
15
|
)
|
|
14
16
|
|
|
15
17
|
|
|
@@ -126,7 +128,7 @@ def test_unknown_spdx_comment_preserves_and_or_from_detected_expression():
|
|
|
126
128
|
success, results, _messages, _ = parsing_scancode(scancode_file_list)
|
|
127
129
|
|
|
128
130
|
assert success is True
|
|
129
|
-
assert results[0].licenses == ["
|
|
131
|
+
assert results[0].licenses == ["DApache-2.0", "GPL-2.0", "NEW"]
|
|
130
132
|
assert "unknown-license-reference" not in [lic.lower() for lic in results[0].licenses]
|
|
131
133
|
assert results[0].comment == "NEW OR DApache-2.0 AND GPL-2.0"
|
|
132
134
|
|
|
@@ -157,6 +159,168 @@ def test_unknown_license_reference_suppressed_when_same_matched_text_has_other_l
|
|
|
157
159
|
assert results[0].licenses == ["GPL-2.0"]
|
|
158
160
|
|
|
159
161
|
|
|
162
|
+
@pytest.mark.parametrize(
|
|
163
|
+
"matched_text",
|
|
164
|
+
[
|
|
165
|
+
"License terms: http://example.com/license",
|
|
166
|
+
"License terms: https://example.com/license",
|
|
167
|
+
"License terms: HTTPS://example.com/license",
|
|
168
|
+
],
|
|
169
|
+
)
|
|
170
|
+
def test_unknown_license_reference_kept_when_it_is_only_license_with_url(matched_text):
|
|
171
|
+
scancode_file_list = [{
|
|
172
|
+
"path": "reference.txt",
|
|
173
|
+
"type": "file",
|
|
174
|
+
"license_detections": [{
|
|
175
|
+
"matches": [{
|
|
176
|
+
"license_expression": "unknown-license-reference",
|
|
177
|
+
"matched_text": matched_text,
|
|
178
|
+
}],
|
|
179
|
+
}],
|
|
180
|
+
"copyrights": [],
|
|
181
|
+
}]
|
|
182
|
+
|
|
183
|
+
success, results, _messages, license_list = parsing_scancode(scancode_file_list)
|
|
184
|
+
|
|
185
|
+
assert success is True
|
|
186
|
+
assert results[0].licenses == ["unknown-license-reference"]
|
|
187
|
+
assert [item.matched_text for item in license_list.values()] == [matched_text]
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
@pytest.mark.parametrize("matched_text", ["See the accompanying license", "", None])
|
|
191
|
+
def test_unknown_license_reference_suppressed_without_url(matched_text):
|
|
192
|
+
scancode_file_list = [{
|
|
193
|
+
"path": "reference.txt",
|
|
194
|
+
"type": "file",
|
|
195
|
+
"license_detections": [{
|
|
196
|
+
"matches": [{
|
|
197
|
+
"license_expression": "unknown-license-reference",
|
|
198
|
+
"matched_text": matched_text,
|
|
199
|
+
}],
|
|
200
|
+
}],
|
|
201
|
+
"copyrights": [],
|
|
202
|
+
}]
|
|
203
|
+
|
|
204
|
+
success, results, _messages, license_list = parsing_scancode(scancode_file_list)
|
|
205
|
+
|
|
206
|
+
assert success is True
|
|
207
|
+
assert results[0].licenses == []
|
|
208
|
+
assert license_list == {}
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
def test_unknown_license_reference_suppressed_by_other_license_in_same_file():
|
|
212
|
+
reference_text = "License terms: https://example.com/license"
|
|
213
|
+
scancode_file_list = [{
|
|
214
|
+
"path": "mixed.txt",
|
|
215
|
+
"type": "file",
|
|
216
|
+
"detected_license_expression": (
|
|
217
|
+
"unknown-license-reference OR mit OR apache-2.0"
|
|
218
|
+
),
|
|
219
|
+
"license_detections": [{
|
|
220
|
+
"matches": [
|
|
221
|
+
{
|
|
222
|
+
"license_expression": "unknown-license-reference",
|
|
223
|
+
"matched_text": reference_text,
|
|
224
|
+
},
|
|
225
|
+
{
|
|
226
|
+
"license_expression": "mit",
|
|
227
|
+
"matched_text": "Permission is hereby granted, free of charge...",
|
|
228
|
+
},
|
|
229
|
+
{
|
|
230
|
+
"license_expression": "apache-2.0",
|
|
231
|
+
"matched_text": "Licensed under the Apache License, Version 2.0",
|
|
232
|
+
},
|
|
233
|
+
],
|
|
234
|
+
}],
|
|
235
|
+
"copyrights": [],
|
|
236
|
+
}]
|
|
237
|
+
|
|
238
|
+
success, results, _messages, license_list = parsing_scancode(scancode_file_list)
|
|
239
|
+
|
|
240
|
+
assert success is True
|
|
241
|
+
assert results[0].licenses == ["MIT", "Apache-2.0"]
|
|
242
|
+
assert results[0].comment == "MIT OR Apache-2.0"
|
|
243
|
+
assert all(
|
|
244
|
+
item.license != "unknown-license-reference"
|
|
245
|
+
for item in license_list.values()
|
|
246
|
+
)
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
def test_unknown_license_reference_filter_is_scoped_to_each_file():
|
|
250
|
+
reference_text = "License terms: https://example.com/license"
|
|
251
|
+
scancode_file_list = [
|
|
252
|
+
{
|
|
253
|
+
"path": "reference.txt",
|
|
254
|
+
"type": "file",
|
|
255
|
+
"license_detections": [{
|
|
256
|
+
"matches": [{
|
|
257
|
+
"license_expression": "unknown-license-reference",
|
|
258
|
+
"matched_text": reference_text,
|
|
259
|
+
}],
|
|
260
|
+
}],
|
|
261
|
+
"copyrights": [],
|
|
262
|
+
},
|
|
263
|
+
{
|
|
264
|
+
"path": "mit.txt",
|
|
265
|
+
"type": "file",
|
|
266
|
+
"license_detections": [{
|
|
267
|
+
"matches": [{
|
|
268
|
+
"license_expression": "mit",
|
|
269
|
+
"matched_text": "Permission is hereby granted, free of charge...",
|
|
270
|
+
}],
|
|
271
|
+
}],
|
|
272
|
+
"copyrights": [],
|
|
273
|
+
},
|
|
274
|
+
]
|
|
275
|
+
|
|
276
|
+
success, results, _messages, _license_list = parsing_scancode(scancode_file_list)
|
|
277
|
+
|
|
278
|
+
assert success is True
|
|
279
|
+
assert results[0].licenses == ["unknown-license-reference"]
|
|
280
|
+
assert results[1].licenses == ["MIT"]
|
|
281
|
+
|
|
282
|
+
|
|
283
|
+
def test_only_url_backed_unknown_license_reference_is_reported():
|
|
284
|
+
valid_text = "License terms: https://example.com/license"
|
|
285
|
+
invalid_text = "See the accompanying license"
|
|
286
|
+
scancode_file_list = [{
|
|
287
|
+
"path": "references.txt",
|
|
288
|
+
"type": "file",
|
|
289
|
+
"license_detections": [{
|
|
290
|
+
"matches": [
|
|
291
|
+
{
|
|
292
|
+
"license_expression": "unknown-license-reference",
|
|
293
|
+
"matched_text": invalid_text,
|
|
294
|
+
},
|
|
295
|
+
{
|
|
296
|
+
"license_expression": "unknown-license-reference",
|
|
297
|
+
"matched_text": valid_text,
|
|
298
|
+
},
|
|
299
|
+
],
|
|
300
|
+
}],
|
|
301
|
+
"copyrights": [],
|
|
302
|
+
}]
|
|
303
|
+
|
|
304
|
+
success, results, _messages, license_list = parsing_scancode(scancode_file_list)
|
|
305
|
+
|
|
306
|
+
assert success is True
|
|
307
|
+
assert results[0].licenses == ["unknown-license-reference"]
|
|
308
|
+
assert [item.matched_text for item in license_list.values()] == [valid_text]
|
|
309
|
+
|
|
310
|
+
|
|
311
|
+
@pytest.mark.parametrize(
|
|
312
|
+
("matched_text", "expected"),
|
|
313
|
+
[
|
|
314
|
+
("http://example.com/license", True),
|
|
315
|
+
("HTTPS://example.com/license", True),
|
|
316
|
+
("ftp://example.com/license", False),
|
|
317
|
+
(None, False),
|
|
318
|
+
],
|
|
319
|
+
)
|
|
320
|
+
def test_matched_text_has_http_url(matched_text, expected):
|
|
321
|
+
assert _matched_text_has_http_url(matched_text) is expected
|
|
322
|
+
|
|
323
|
+
|
|
160
324
|
def test_licenseref_tokens_stripped_from_unknown_spdx_and_expression():
|
|
161
325
|
scancode_file_list = [{
|
|
162
326
|
"path": "refs.py",
|
|
@@ -196,11 +360,11 @@ def test_build_comment_from_detected_expression_helper():
|
|
|
196
360
|
"matched_text": matched_same,
|
|
197
361
|
},
|
|
198
362
|
]
|
|
199
|
-
|
|
363
|
+
has_other_license = _file_has_other_license(matches)
|
|
200
364
|
comment = build_comment_from_detected_expression(
|
|
201
365
|
"(unknown-spdx OR unknown-spdx) AND unknown-license-reference AND gpl-2.0",
|
|
202
366
|
matches,
|
|
203
|
-
|
|
367
|
+
_should_suppress_unknown_license_reference(matches, has_other_license),
|
|
204
368
|
)
|
|
205
369
|
assert comment == "NEW OR DApache-2.0 AND GPL-2.0"
|
|
206
370
|
|
|
@@ -216,7 +380,7 @@ def test_comment_without_parens_uses_operator_before_kept_token():
|
|
|
216
380
|
comment = build_comment_from_detected_expression(
|
|
217
381
|
"mit OR unknown-license-reference AND apache-2.0",
|
|
218
382
|
matches,
|
|
219
|
-
|
|
383
|
+
True,
|
|
220
384
|
)
|
|
221
385
|
assert comment == "MIT AND Apache-2.0"
|
|
222
386
|
|
|
@@ -232,7 +396,7 @@ def test_comment_with_parens_preserves_or_group():
|
|
|
232
396
|
comment = build_comment_from_detected_expression(
|
|
233
397
|
"mit OR (unknown-license-reference AND apache-2.0)",
|
|
234
398
|
matches,
|
|
235
|
-
|
|
399
|
+
True,
|
|
236
400
|
)
|
|
237
401
|
assert comment == "MIT OR Apache-2.0"
|
|
238
402
|
|
|
@@ -248,7 +412,7 @@ def test_comment_with_parens_preserves_and_after_group():
|
|
|
248
412
|
comment = build_comment_from_detected_expression(
|
|
249
413
|
"(mit OR unknown-license-reference) AND apache-2.0",
|
|
250
414
|
matches,
|
|
251
|
-
|
|
415
|
+
True,
|
|
252
416
|
)
|
|
253
417
|
assert comment == "MIT AND Apache-2.0"
|
|
254
418
|
|
|
@@ -337,10 +501,10 @@ def test_android_bp_soong_license_kinds_without_line_comment_in_license():
|
|
|
337
501
|
licenses = results[0].licenses
|
|
338
502
|
assert licenses == [
|
|
339
503
|
"Apache-2.0",
|
|
340
|
-
"unknown-license-reference",
|
|
341
504
|
"BSD",
|
|
342
505
|
"MIT",
|
|
343
506
|
"OFL",
|
|
507
|
+
"unknown-license-reference",
|
|
344
508
|
]
|
|
345
509
|
assert all("//" not in lic for lic in results[0].licenses)
|
|
346
510
|
assert all('"' not in lic for lic in results[0].licenses)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{fosslight_source-2.3.9 → fosslight_source-2.3.11}/src/fosslight_source/run_manifest_extractor.py
RENAMED
|
File without changes
|
|
File without changes
|
{fosslight_source-2.3.9 → fosslight_source-2.3.11}/src/fosslight_source/run_spdx_extractor.py
RENAMED
|
File without changes
|
|
File without changes
|
{fosslight_source-2.3.9 → fosslight_source-2.3.11}/src/fosslight_source.egg-info/entry_points.txt
RENAMED
|
File without changes
|
{fosslight_source-2.3.9 → fosslight_source-2.3.11}/src/fosslight_source.egg-info/top_level.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|