fosslight-source 2.3.9__tar.gz → 2.3.11__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. {fosslight_source-2.3.9/src/fosslight_source.egg-info → fosslight_source-2.3.11}/PKG-INFO +3 -2
  2. {fosslight_source-2.3.9 → fosslight_source-2.3.11}/pyproject.toml +3 -2
  3. fosslight_source-2.3.11/src/fosslight_source/_exclude.py +21 -0
  4. {fosslight_source-2.3.9 → fosslight_source-2.3.11}/src/fosslight_source/_help.py +2 -0
  5. {fosslight_source-2.3.9 → fosslight_source-2.3.11}/src/fosslight_source/_kb_client.py +28 -1
  6. {fosslight_source-2.3.9 → fosslight_source-2.3.11}/src/fosslight_source/_merge.py +2 -1
  7. {fosslight_source-2.3.9 → fosslight_source-2.3.11}/src/fosslight_source/_parsing_scancode_file_item.py +69 -30
  8. {fosslight_source-2.3.9 → fosslight_source-2.3.11}/src/fosslight_source/_parsing_scanoss_file.py +2 -1
  9. {fosslight_source-2.3.9 → fosslight_source-2.3.11}/src/fosslight_source/_scan_item.py +6 -4
  10. {fosslight_source-2.3.9 → fosslight_source-2.3.11}/src/fosslight_source/_scancode_ignore_binaries.py +43 -8
  11. {fosslight_source-2.3.9 → fosslight_source-2.3.11}/src/fosslight_source/cli.py +9 -5
  12. {fosslight_source-2.3.9 → fosslight_source-2.3.11}/src/fosslight_source/run_scancode.py +42 -1
  13. {fosslight_source-2.3.9 → fosslight_source-2.3.11/src/fosslight_source.egg-info}/PKG-INFO +3 -2
  14. {fosslight_source-2.3.9 → fosslight_source-2.3.11}/src/fosslight_source.egg-info/SOURCES.txt +3 -0
  15. {fosslight_source-2.3.9 → fosslight_source-2.3.11}/src/fosslight_source.egg-info/requires.txt +2 -1
  16. fosslight_source-2.3.11/tests/test_kb_ssl.py +52 -0
  17. {fosslight_source-2.3.9 → fosslight_source-2.3.11}/tests/test_manifest_android_bp.py +1 -1
  18. {fosslight_source-2.3.9 → fosslight_source-2.3.11}/tests/test_manifest_recommended_scenarios.py +1 -1
  19. fosslight_source-2.3.11/tests/test_multi_value_order.py +26 -0
  20. {fosslight_source-2.3.9 → fosslight_source-2.3.11}/tests/test_parsing_unknown_spdx.py +172 -8
  21. {fosslight_source-2.3.9 → fosslight_source-2.3.11}/LICENSE +0 -0
  22. {fosslight_source-2.3.9 → fosslight_source-2.3.11}/MANIFEST.in +0 -0
  23. {fosslight_source-2.3.9 → fosslight_source-2.3.11}/README.md +0 -0
  24. {fosslight_source-2.3.9 → fosslight_source-2.3.11}/setup.cfg +0 -0
  25. {fosslight_source-2.3.9 → fosslight_source-2.3.11}/src/fosslight_source/__init__.py +0 -0
  26. {fosslight_source-2.3.9 → fosslight_source-2.3.11}/src/fosslight_source/_license_matched.py +0 -0
  27. {fosslight_source-2.3.9 → fosslight_source-2.3.11}/src/fosslight_source/run_manifest_extractor.py +0 -0
  28. {fosslight_source-2.3.9 → fosslight_source-2.3.11}/src/fosslight_source/run_scanoss.py +0 -0
  29. {fosslight_source-2.3.9 → fosslight_source-2.3.11}/src/fosslight_source/run_spdx_extractor.py +0 -0
  30. {fosslight_source-2.3.9 → fosslight_source-2.3.11}/src/fosslight_source.egg-info/dependency_links.txt +0 -0
  31. {fosslight_source-2.3.9 → fosslight_source-2.3.11}/src/fosslight_source.egg-info/entry_points.txt +0 -0
  32. {fosslight_source-2.3.9 → fosslight_source-2.3.11}/src/fosslight_source.egg-info/top_level.txt +0 -0
  33. {fosslight_source-2.3.9 → fosslight_source-2.3.11}/tests/test_manifest_composer.py +0 -0
  34. {fosslight_source-2.3.9 → fosslight_source-2.3.11}/tests/test_manifest_pyproject.py +0 -0
  35. {fosslight_source-2.3.9 → fosslight_source-2.3.11}/tests/test_tox.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: fosslight_source
3
- Version: 2.3.9
3
+ Version: 2.3.11
4
4
  Summary: FOSSLight Source Scanner
5
5
  Author: LG Electronics
6
6
  License-Expression: Apache-2.0
@@ -20,7 +20,7 @@ Requires-Dist: setuptools<=80.10.2
20
20
  Requires-Dist: pyparsing
21
21
  Requires-Dist: scanoss>=1.45.0
22
22
  Requires-Dist: XlsxWriter
23
- Requires-Dist: fosslight_util>=2.2.8
23
+ Requires-Dist: fosslight_util>=2.2.12
24
24
  Requires-Dist: PyYAML
25
25
  Requires-Dist: wheel>=0.38.1
26
26
  Requires-Dist: intbitset
@@ -32,6 +32,7 @@ Requires-Dist: normality==2.6.1
32
32
  Requires-Dist: psycopg2-binary>=2.9.10; python_version >= "3.13"
33
33
  Requires-Dist: tomli; python_version < "3.11"
34
34
  Requires-Dist: tqdm
35
+ Requires-Dist: truststore
35
36
  Dynamic: license-file
36
37
 
37
38
  <!--
@@ -7,7 +7,7 @@ build-backend = "setuptools.build_meta"
7
7
 
8
8
  [project]
9
9
  name = "fosslight_source"
10
- version = "2.3.9"
10
+ version = "2.3.11"
11
11
  description = "FOSSLight Source Scanner"
12
12
  readme = "README.md"
13
13
  license = "Apache-2.0"
@@ -29,7 +29,7 @@ dependencies = [
29
29
  "pyparsing",
30
30
  "scanoss>=1.45.0",
31
31
  "XlsxWriter",
32
- "fosslight_util>=2.2.8",
32
+ "fosslight_util>=2.2.12",
33
33
  "PyYAML",
34
34
  "wheel>=0.38.1",
35
35
  "intbitset",
@@ -43,6 +43,7 @@ dependencies = [
43
43
  "psycopg2-binary>=2.9.10; python_version >= '3.13'",
44
44
  "tomli; python_version < '3.11'",
45
45
  "tqdm",
46
+ "truststore",
46
47
  ]
47
48
 
48
49
  [project.optional-dependencies]
@@ -0,0 +1,21 @@
1
+ # Copyright (c) 2026 LG Electronics Inc.
2
+ # SPDX-License-Identifier: Apache-2.0
3
+
4
+ """Source-scanner filename excludes (not applied to dependency discovery)."""
5
+
6
+ from fosslight_util.exclude import is_excluded_filename
7
+
8
+ # Build/package config noise for license scans.
9
+ EXCLUDE_FILENAME_SOURCE = frozenset({
10
+ "changelog", "config.guess", "config.sub", "changes", "ltmain.sh",
11
+ "configure", "configure.ac", "depcomp", "compile", "missing", "makefile",
12
+ "makefile.am",
13
+ "build.gradle", "build.gradle.kts", "settings.gradle", "settings.gradle.kts",
14
+ "gradlew", "gradlew.bat",
15
+ "vite.config.ts", "vite.config.js", "vite.config.mts", "vite.config.mjs",
16
+ "package-lock.json", "npm-shrinkwrap.json", "yarn.lock", "pnpm-lock.yaml",
17
+ })
18
+
19
+
20
+ def is_excluded_source_filename(file_path: str) -> bool:
21
+ return is_excluded_filename(file_path, EXCLUDE_FILENAME_SOURCE)
@@ -45,6 +45,8 @@ _HELP_MESSAGE_SOURCE_SCANNER = f"""
45
45
  --hide_progress Hide the progress bar during scanning
46
46
  --kb_url <url> KB API URL (priority: parameter > KB_URL env > default)
47
47
  --kb_token <token> KB bearer token (priority: parameter > KB_TOKEN env)
48
+ HTTPS uses the OS certificate store. Set KB_SSL_VERIFY=false
49
+ only if certificate verification must be skipped.
48
50
 
49
51
  💡 Examples
50
52
  ────────────────────────────────────────────────────────────────────
@@ -5,6 +5,8 @@
5
5
 
6
6
  import json
7
7
  import logging
8
+ import os
9
+ import ssl
8
10
  import time
9
11
  import urllib.error
10
12
  import urllib.request
@@ -19,6 +21,31 @@ _SCAN_JOB_POLL_MAX_INTERVAL_SEC = 10.0
19
21
  _SCAN_JOB_REQUEST_TIMEOUT_SEC = 30
20
22
  _SCAN_JOB_MIN_WAIT_SEC = 300
21
23
  _SCAN_JOB_PER_HASH_SEC = 35
24
+ _KB_SSL_VERIFY_FALSE = {"0", "false", "no", "off"}
25
+ _logged_insecure_kb_ssl = False
26
+
27
+
28
+ def kb_ssl_verify_enabled() -> bool:
29
+ raw = os.environ.get("KB_SSL_VERIFY", "true")
30
+ return raw.strip().lower() not in _KB_SSL_VERIFY_FALSE
31
+
32
+
33
+ def create_kb_ssl_context() -> ssl.SSLContext:
34
+ """TLS context for KB HTTPS. Uses the OS trust store (Windows store, like the browser)."""
35
+ global _logged_insecure_kb_ssl
36
+ if not kb_ssl_verify_enabled():
37
+ if not _logged_insecure_kb_ssl:
38
+ logger.warning("KB TLS certificate verification is disabled (KB_SSL_VERIFY)")
39
+ _logged_insecure_kb_ssl = True
40
+ ctx = ssl.SSLContext(ssl.PROTOCOL_TLS_CLIENT)
41
+ ctx.check_hostname = False
42
+ ctx.verify_mode = ssl.CERT_NONE
43
+ return ctx
44
+ try:
45
+ import truststore
46
+ return truststore.SSLContext(ssl.PROTOCOL_TLS_CLIENT)
47
+ except ImportError:
48
+ return ssl.create_default_context()
22
49
 
23
50
 
24
51
  def _kb_request(
@@ -40,7 +67,7 @@ def _kb_request(
40
67
  if kb_token:
41
68
  request.add_header("Authorization", f"Bearer {kb_token}")
42
69
 
43
- with urllib.request.urlopen(request, timeout=timeout) as response:
70
+ with urllib.request.urlopen(request, timeout=timeout, context=create_kb_ssl_context()) as response:
44
71
  body = response.read().decode()
45
72
  return json.loads(body) if body else {}
46
73
 
@@ -111,7 +111,8 @@ def _get_top_merge_values(scan_items: list, value_getter) -> list:
111
111
  normalized_value = _normalize_merge_text(value)
112
112
  if normalized_value:
113
113
  values.append(normalized_value)
114
- return [value for value, _ in Counter(values).most_common(3)]
114
+ ranked = sorted(Counter(values).items(), key=lambda entry: (-entry[1], entry[0]))
115
+ return [value for value, _ in ranked[:3]]
115
116
 
116
117
 
117
118
  def _can_merge_folder(scan_items: list) -> bool:
@@ -6,8 +6,9 @@
6
6
  import os
7
7
  import logging
8
8
  import re
9
+ from functools import lru_cache
9
10
  import fosslight_util.constant as constant
10
- from fosslight_util.exclude import is_excluded_filename
11
+ from ._exclude import is_excluded_source_filename
11
12
  from ._license_matched import MatchedLicense
12
13
  from ._scan_item import SourceItem
13
14
  from ._scan_item import replace_word
@@ -34,6 +35,7 @@ KEYWORD_SPDX_ID = r'SPDX-License-Identifier\s*[\S]+'
34
35
  KEYWORD_DOWNLOAD_LOC = r'DownloadLocation\s*[\S]+'
35
36
  KEYWORD_SCANCODE_UNKNOWN = "unknown-spdx"
36
37
  KEYWORD_UNKNOWN_LICENSE_REFERENCE = "unknown-license-reference"
38
+ HTTP_URL_PATTERN = re.compile(r'https?://', re.IGNORECASE)
37
39
  SPDX_REPLACE_WORDS = ["(", ")"]
38
40
  KEY_AND_OR = re.compile(r"(?<=\s)(?:and|or)(?=\s)", re.IGNORECASE)
39
41
  KEY_AND_OR_CAPTURE = re.compile(r"(?<=\s)(and|or)(?=\s)", re.IGNORECASE)
@@ -61,25 +63,41 @@ def filter_fsf_copyright_from_gpl_license_text(
61
63
  return [c for c in copyrights if FSF_IN_COPYRIGHT not in c.lower()]
62
64
 
63
65
 
64
- def _expression_has_non_unknown_license_reference(license_expression: str) -> bool:
66
+ def _expression_has_other_license(license_expression: str) -> bool:
65
67
  if not license_expression:
66
68
  return False
67
69
  for token in split_spdx_expression(license_expression.lower()):
68
70
  token = token.strip()
69
- if token and KEYWORD_UNKNOWN_LICENSE_REFERENCE not in token:
71
+ if (
72
+ token
73
+ and KEYWORD_UNKNOWN_LICENSE_REFERENCE not in token
74
+ and token not in REMOVE_LICENSE
75
+ ):
70
76
  return True
71
77
  return False
72
78
 
73
79
 
74
- def _matched_texts_with_other_licenses(matches: list) -> set:
75
- """matched_text values that also produced a non-unknown-license-reference license."""
76
- texts = set()
80
+ def _file_has_other_license(matches: list) -> bool:
81
+ """Return whether a file has a license other than unknown-license-reference."""
77
82
  for matched_lic in matches or []:
78
- matched_txt = matched_lic.get("matched_text") or ""
79
83
  license_expression = matched_lic.get("license_expression") or ""
80
- if matched_txt and _expression_has_non_unknown_license_reference(license_expression):
81
- texts.add(matched_txt)
82
- return texts
84
+ if _expression_has_other_license(license_expression):
85
+ return True
86
+ return False
87
+
88
+
89
+ def _matched_text_has_http_url(matched_text: str) -> bool:
90
+ """Return whether matched text contains an HTTP or HTTPS URL."""
91
+ return bool(HTTP_URL_PATTERN.search(matched_text or ""))
92
+
93
+
94
+ def _should_keep_unknown_license_reference(
95
+ matched_text: str, has_other_license_in_file: bool
96
+ ) -> bool:
97
+ return (
98
+ not has_other_license_in_file
99
+ and _matched_text_has_http_url(matched_text)
100
+ )
83
101
 
84
102
 
85
103
  def get_error_from_header(header_item: list) -> Tuple[bool, str]:
@@ -424,23 +442,25 @@ def _build_unknown_spdx_replacement_queue(matches: list) -> list[str]:
424
442
 
425
443
 
426
444
  def _should_suppress_unknown_license_reference(
427
- matches: list, matched_texts_with_other_licenses: set
445
+ matches: list, has_other_license_in_file: bool
428
446
  ) -> bool:
447
+ has_unknown_license_reference = False
429
448
  for matched_lic in matches or []:
430
449
  expr = (matched_lic.get("license_expression") or "").lower()
431
- matched_txt = matched_lic.get("matched_text") or ""
432
- if (
433
- KEYWORD_UNKNOWN_LICENSE_REFERENCE in expr
434
- and matched_txt in matched_texts_with_other_licenses
450
+ if KEYWORD_UNKNOWN_LICENSE_REFERENCE not in expr:
451
+ continue
452
+ has_unknown_license_reference = True
453
+ if _should_keep_unknown_license_reference(
454
+ matched_lic.get("matched_text") or "", has_other_license_in_file
435
455
  ):
436
- return True
437
- return False
456
+ return False
457
+ return has_unknown_license_reference
438
458
 
439
459
 
440
460
  def build_comment_from_detected_expression(
441
461
  detected_expression: str,
442
462
  matches: list,
443
- matched_texts_with_other_licenses: set,
463
+ suppress_unknown_license_reference: bool,
444
464
  ) -> str:
445
465
  """
446
466
  Rebuild comment from detected expression.
@@ -470,26 +490,35 @@ def build_comment_from_detected_expression(
470
490
  )
471
491
 
472
492
  replacements = _build_unknown_spdx_replacement_queue(matches)
473
- suppress_ulr = _should_suppress_unknown_license_reference(
474
- matches, matched_texts_with_other_licenses
475
- )
476
493
  tree = _parse_license_expression_tokens(_tokenize_license_expression(expr))
477
494
  if tree is None:
478
495
  return ""
479
496
  transformed = _transform_license_expr_node(
480
- tree, replacements, [0], suppress_ulr
497
+ tree, replacements, [0], suppress_unknown_license_reference
481
498
  )
482
499
  return _omit_parens_for_two_license_expression(
483
500
  _serialize_license_expr_node(transformed)
484
501
  )
485
502
 
486
503
 
504
+ @lru_cache(maxsize=65536)
487
505
  def get_license_expression_spdx(license_expression: str) -> str:
488
506
  if not license_expression or not license_expression.strip():
489
507
  return ""
490
508
  try:
491
- from licensedcode.cache import build_spdx_license_expression
492
- result = build_spdx_license_expression(license_expression.strip())
509
+ from licensedcode.cache import (
510
+ build_spdx_license_expression,
511
+ get_licenses_db,
512
+ get_licensing,
513
+ )
514
+ expression = license_expression.strip()
515
+ # licensedcode re-reads the whole license database from disk for every
516
+ # key it cannot resolve, and then raises anyway. Screen the keys first
517
+ # so unknown tokens stay cheap.
518
+ licenses_db = get_licenses_db()
519
+ if any(key not in licenses_db for key in get_licensing().license_keys(expression)):
520
+ return ""
521
+ result = build_spdx_license_expression(expression)
493
522
  if result is None:
494
523
  return ""
495
524
  if isinstance(result, str) and result.lower().startswith("licenseref-"):
@@ -516,7 +545,7 @@ def parsing_scancode(
516
545
  if (not file_path) or is_binary or is_dir:
517
546
  logger.info(f"Skipping {file_path} because it is binary or directory")
518
547
  continue
519
- if is_excluded_filename(file_path):
548
+ if is_excluded_source_filename(file_path):
520
549
  logger.debug(f"Skipping {file_path} because it is an excluded filename")
521
550
  continue
522
551
  result_item = SourceItem(file_path)
@@ -545,7 +574,12 @@ def parsing_scancode(
545
574
  all_matches = []
546
575
  for lic in licenses or []:
547
576
  all_matches.extend(lic.get("matches") or [])
548
- matched_texts_with_other_licenses = _matched_texts_with_other_licenses(all_matches)
577
+ has_other_license_in_file = _file_has_other_license(all_matches)
578
+ suppress_unknown_license_reference = (
579
+ _should_suppress_unknown_license_reference(
580
+ all_matches, has_other_license_in_file
581
+ )
582
+ )
549
583
  for lic in licenses or []:
550
584
  matched_lic_list = lic.get("matches", [])
551
585
  for matched_lic in matched_lic_list:
@@ -565,7 +599,9 @@ def parsing_scancode(
565
599
  continue
566
600
  if (
567
601
  KEYWORD_UNKNOWN_LICENSE_REFERENCE in found_lic.lower()
568
- and matched_txt in matched_texts_with_other_licenses
602
+ and not _should_keep_unknown_license_reference(
603
+ matched_txt, has_other_license_in_file
604
+ )
569
605
  ):
570
606
  continue
571
607
  if KEYWORD_SCANCODE_UNKNOWN in found_lic.lower():
@@ -592,8 +628,10 @@ def parsing_scancode(
592
628
  file.get("percentage_of_license_text", 0) > 90 and not is_source_file
593
629
  )
594
630
 
595
- result_item.copyright = filter_fsf_copyright_from_gpl_license_text(
596
- copyright_value_list, license_detected, result_item.is_license_text
631
+ result_item.copyright = sorted(
632
+ filter_fsf_copyright_from_gpl_license_text(
633
+ copyright_value_list, license_detected, result_item.is_license_text
634
+ )
597
635
  )
598
636
 
599
637
  if len(license_detected) > 1:
@@ -603,6 +641,7 @@ def parsing_scancode(
603
641
  resolved_unknown_spdx
604
642
  or KEYWORD_SCANCODE_UNKNOWN in detected_expression.lower()
605
643
  or "licenseref-scancode-unknown-spdx" in detected_expression_spdx.lower()
644
+ or suppress_unknown_license_reference
606
645
  ):
607
646
  # Prefer non-SPDX expression so unknown-spdx tokens map cleanly.
608
647
  # Comment only for dual-license style expressions that include OR.
@@ -611,7 +650,7 @@ def parsing_scancode(
611
650
  result_item.comment = build_comment_from_detected_expression(
612
651
  source_expression,
613
652
  all_matches,
614
- matched_texts_with_other_licenses,
653
+ suppress_unknown_license_reference,
615
654
  )
616
655
  else:
617
656
  license_expression = detected_expression_spdx or detected_expression
@@ -5,6 +5,7 @@
5
5
 
6
6
  import logging
7
7
  import fosslight_util.constant as constant
8
+ from ._exclude import is_excluded_source_filename
8
9
  from ._scan_item import SourceItem
9
10
  from ._scan_item import replace_word
10
11
  from typing import Tuple
@@ -39,7 +40,7 @@ def parsing_scan_result(scanoss_report: dict, excluded_files: set = None) -> Tup
39
40
 
40
41
  for file_path, findings in scanoss_report.items():
41
42
  file_path_normalized = file_path.replace('\\', '/')
42
- if file_path_normalized in excluded_files:
43
+ if file_path_normalized in excluded_files or is_excluded_source_filename(file_path_normalized):
43
44
  continue
44
45
  result_item = SourceItem(file_path)
45
46
 
@@ -7,6 +7,7 @@ import os
7
7
  import logging
8
8
  import re
9
9
  import hashlib
10
+ from bisect import insort
10
11
  import fosslight_util.constant as constant
11
12
  from fosslight_util.oss_item import FileItem, OssItem, get_checksum_sha1
12
13
 
@@ -78,13 +79,13 @@ class SourceItem(FileItem):
78
79
  def licenses(self, value: list) -> None:
79
80
  if value:
80
81
  max_length_exceed = False
81
- for new_lic in value:
82
+ for new_lic in sorted(value):
82
83
  if new_lic:
83
84
  if len(new_lic) > MAX_LICENSE_LENGTH:
84
85
  new_lic = new_lic[:MAX_LICENSE_LENGTH]
85
86
  max_length_exceed = True
86
87
  if new_lic not in self._licenses:
87
- self._licenses.append(new_lic)
88
+ insort(self._licenses, new_lic)
88
89
  if len(",".join(self._licenses)) > MAX_LICENSE_TOTAL_LENGTH:
89
90
  self._licenses.remove(new_lic)
90
91
  max_length_exceed = True
@@ -180,10 +181,11 @@ class SourceItem(FileItem):
180
181
  kb_origin_urls: dict[str, str] | None = None,
181
182
  ) -> None:
182
183
  self.oss_items = []
184
+ copyrights = "\n".join(self.copyright)
183
185
  if self.download_location:
184
186
  for url in self.download_location:
185
187
  item = OssItem(self.oss_name, self.oss_version, self.licenses, url)
186
- item.copyright = "\n".join(self.copyright)
188
+ item.copyright = copyrights
187
189
  item.comment = self.comment
188
190
  self.oss_items.append(item)
189
191
  else:
@@ -198,7 +200,7 @@ class SourceItem(FileItem):
198
200
  oss_name, oss_version, download_url = self._apply_kb_origin_url(origin_url)
199
201
  item = OssItem(oss_name, oss_version, self.licenses, download_url)
200
202
 
201
- item.copyright = "\n".join(self.copyright)
203
+ item.copyright = copyrights
202
204
  item.comment = self.comment
203
205
  self.oss_items.append(item)
204
206
 
@@ -6,12 +6,22 @@
6
6
  # so PyPI installs do not need a GitHub git dependency.
7
7
  # SPDX-PackageDownloadLocation: https://github.com/aboutcode-org/scancode-plugins/tree/main/misc/scancode-ignore-binaries
8
8
 
9
+ import logging
10
+ import multiprocessing
11
+
9
12
  from plugincode.pre_scan import PreScanPlugin
10
13
  from plugincode.pre_scan import pre_scan_impl
11
14
  from commoncode.cliutils import PluggableCommandLineOption
12
15
  from commoncode.cliutils import PRE_SCAN_GROUP
13
16
  from typecode.contenttype import get_type
14
17
 
18
+ logger = logging.getLogger(__name__)
19
+
20
+ # Detection costs one get_type() call per file, which reads from disk. On large
21
+ # trees that dominates the pre-scan stage, so spread it over the scan processes.
22
+ PARALLEL_MIN_FILES = 2000
23
+ CHUNK_SIZE = 256
24
+
15
25
 
16
26
  @pre_scan_impl
17
27
  class IgnoreBinaries(PreScanPlugin):
@@ -32,22 +42,47 @@ class IgnoreBinaries(PreScanPlugin):
32
42
  def is_enabled(self, ignore_binaries, **kwargs):
33
43
  return ignore_binaries
34
44
 
35
- def process_codebase(self, codebase, ignore_binaries, **kwargs):
45
+ def process_codebase(self, codebase, ignore_binaries, processes=1, **kwargs):
36
46
  """
37
47
  Remove binary Resources from the resource tree.
38
48
  """
39
49
  if not ignore_binaries:
40
50
  return
41
51
 
42
- resources_to_remove = []
43
- for resource in codebase.walk():
44
- if not resource.is_file:
52
+ # Collect first: removing while walking would invalidate the walk.
53
+ candidates = [
54
+ (resource.path, resource.location)
55
+ for resource in codebase.walk()
56
+ if resource.is_file
57
+ ]
58
+ if not candidates:
59
+ return
60
+
61
+ locations = [location for _path, location in candidates]
62
+ flags = _detect_binaries(locations, processes)
63
+
64
+ for (path, _location), binary in zip(candidates, flags):
65
+ if not binary:
45
66
  continue
46
- if is_binary(resource.location):
47
- resources_to_remove.append(resource)
67
+ resource = codebase.get_resource(path)
68
+ if resource is not None:
69
+ resource.remove(codebase)
70
+
71
+
72
+ def _detect_binaries(locations, processes):
73
+ """
74
+ Return a list of booleans, one per location, telling whether it is binary.
75
+ """
76
+ if processes and processes > 1 and len(locations) >= PARALLEL_MIN_FILES:
77
+ try:
78
+ pool = multiprocessing.Pool(processes)
79
+ except Exception as ex:
80
+ logger.debug(f"Parallel binary detection unavailable, using one process: {ex}")
81
+ else:
82
+ with pool:
83
+ return pool.map(is_binary, locations, chunksize=CHUNK_SIZE)
48
84
 
49
- for resource in resources_to_remove:
50
- resource.remove(codebase)
85
+ return [is_binary(location) for location in locations]
51
86
 
52
87
 
53
88
  def is_binary(location):
@@ -22,6 +22,7 @@ from fosslight_util.correct import correct_with_yaml
22
22
  from fosslight_util.parsing_yaml import SUPPORT_OSS_INFO_FILES
23
23
  from .run_scancode import run_scan
24
24
  from fosslight_util.exclude import get_excluded_paths
25
+ from ._exclude import EXCLUDE_FILENAME_SOURCE, is_excluded_source_filename
25
26
  from .run_scanoss import run_scanoss_py
26
27
  from .run_scanoss import get_scanoss_extra_info
27
28
  import yaml
@@ -30,7 +31,7 @@ import argparse
30
31
  from .run_spdx_extractor import get_spdx_downloads
31
32
  from .run_manifest_extractor import get_manifest_licenses
32
33
  from ._scan_item import SourceItem, resolve_kb_config, is_notice_file, is_manifest_file
33
- from ._kb_client import fetch_origin_urls_via_scan_job
34
+ from ._kb_client import create_kb_ssl_context, fetch_origin_urls_via_scan_job
34
35
  from fosslight_util.cover import dump_result_log
35
36
  from fosslight_util.time import current_timestamp_utc, format_running_time, timestamp_for_filename
36
37
  from fosslight_util.oss_item import ScannerItem
@@ -295,7 +296,7 @@ def check_kb_server_reachable(kb_url: str, kb_token: str = "") -> bool:
295
296
  request = urllib.request.Request(f"{kb_url}health", method='GET')
296
297
  if kb_token:
297
298
  request.add_header('Authorization', f'Bearer {kb_token}')
298
- with urllib.request.urlopen(request, timeout=10) as response:
299
+ with urllib.request.urlopen(request, timeout=10, context=create_kb_ssl_context()) as response:
299
300
  logger.debug(f"KB server is reachable. Response status: {response.status}")
300
301
  return True
301
302
  except urllib.error.HTTPError:
@@ -380,7 +381,8 @@ def _collect_kb_file_hashes(
380
381
 
381
382
  for file_path in tqdm.tqdm(files_to_scan, desc="KB Hashing", disable=hide_progress):
382
383
  rel_path = os.path.relpath(file_path, abs_path_to_scan).replace("\\", "/")
383
- if rel_path in scancode_paths or rel_path in excluded_files or is_notice_file(file_path):
384
+ if (rel_path in scancode_paths or rel_path in excluded_files
385
+ or is_excluded_source_filename(rel_path) or is_notice_file(file_path)):
384
386
  continue
385
387
  extra_item = SourceItem(rel_path)
386
388
  md5_hash, _wfp = extra_item._get_hash(path_to_scan)
@@ -629,7 +631,9 @@ def run_scanners(
629
631
  (excluded_path_with_default_exclusion,
630
632
  excluded_path_without_dot,
631
633
  excluded_files,
632
- cnt_file_except_skipped) = get_excluded_paths(path_to_scan, path_to_exclude_with_filename)
634
+ cnt_file_except_skipped) = get_excluded_paths(
635
+ path_to_scan, path_to_exclude_with_filename,
636
+ exclude_filenames=EXCLUDE_FILENAME_SOURCE)
633
637
  logger.debug(f"Skipped paths count: {len(excluded_path_with_default_exclusion)}")
634
638
 
635
639
  if not selected_scanner:
@@ -727,7 +731,7 @@ def metadata_collector(path_to_scan: str, excluded_files: set) -> tuple[dict, di
727
731
  for file in files:
728
732
  file_path = os.path.join(root, file)
729
733
  rel_path_file = os.path.relpath(file_path, abs_path_to_scan).replace('\\', '/')
730
- if rel_path_file in excluded_files:
734
+ if rel_path_file in excluded_files or is_excluded_source_filename(rel_path_file):
731
735
  continue
732
736
 
733
737
  downloads = get_spdx_downloads(file_path)
@@ -26,6 +26,11 @@ logger = logging.getLogger(constant.LOGGER_NAME)
26
26
  warnings.filterwarnings("ignore", category=FutureWarning)
27
27
  _PKG_NAME = "fosslight_source"
28
28
 
29
+ _MAX_IN_MEMORY_ENV = "FOSSLIGHT_SCANCODE_MAX_IN_MEMORY"
30
+ _SCANCODE_DEFAULT_MAX_IN_MEMORY = 10000
31
+ _MEMORY_FRACTION_FOR_CODEBASE = 0.4
32
+ _BYTES_PER_RESOURCE = 4096
33
+
29
34
  try:
30
35
  from click.core import UNSET as _CLICK_UNSET # Click >= 8.3
31
36
  _HAS_CLICK_UNSET = True
@@ -165,7 +170,7 @@ def _default_scancode_ignore_patterns(
165
170
  Directory names use path-based globs (e.g. **/tests/**) so they do not match
166
171
  the scan root directory name itself.
167
172
  Binary files are excluded separately via scancode --ignore-binaries.
168
- EXCLUDE_FILENAME is not passed to ScanCode --ignore: matching many exact
173
+ EXCLUDE_FILENAME_SOURCE is not passed to ScanCode --ignore: matching many exact
169
174
  names on a large tree is slow. Those files are dropped after parsing.
170
175
  """
171
176
  patterns = {".*"}
@@ -182,6 +187,39 @@ def _default_scancode_ignore_patterns(
182
187
  return tuple(sorted(patterns))
183
188
 
184
189
 
190
+ def _available_memory_bytes() -> int:
191
+ try:
192
+ with open("/proc/meminfo") as meminfo:
193
+ for line in meminfo:
194
+ if line.startswith("MemAvailable:"):
195
+ return int(line.split()[1]) * 1024
196
+ except OSError:
197
+ pass
198
+ try:
199
+ return os.sysconf("SC_PAGE_SIZE") * os.sysconf("SC_AVPHYS_PAGES")
200
+ except (AttributeError, ValueError, OSError):
201
+ return 0
202
+
203
+
204
+ def _resolve_max_in_memory() -> int:
205
+ """
206
+ Past --max-in-memory resources, ScanCode caches every Resource in its own
207
+ file under the temp dir, which makes the inventory and every codebase walk
208
+ disk bound. Raise the threshold to whatever memory allows and let ScanCode
209
+ keep spilling beyond that.
210
+ """
211
+ override = os.environ.get(_MAX_IN_MEMORY_ENV, "").strip()
212
+ if override:
213
+ try:
214
+ return int(override)
215
+ except ValueError:
216
+ logger.warning(f"Ignoring invalid {_MAX_IN_MEMORY_ENV}={override}")
217
+
218
+ budget = _available_memory_bytes() * _MEMORY_FRACTION_FOR_CODEBASE
219
+ resolved = int(budget // _BYTES_PER_RESOURCE)
220
+ return max(resolved, _SCANCODE_DEFAULT_MAX_IN_MEMORY)
221
+
222
+
185
223
  def run_scan(
186
224
  path_to_scan: str, output_file_name: str = "",
187
225
  _write_json_file: bool = False, num_cores: int = -1,
@@ -252,9 +290,12 @@ def run_scan(
252
290
  path_to_exclude, abs_path_to_scan
253
291
  )
254
292
  logger.debug(f"Scancode ignore patterns: {len(ignore_tuple)}")
293
+ max_in_memory = _resolve_max_in_memory()
294
+ logger.debug(f"Scancode max_in_memory: {max_in_memory}")
255
295
 
256
296
  kwargs = {
257
297
  "max_depth": 100,
298
+ "max_in_memory": max_in_memory,
258
299
  "strip_root": True,
259
300
  "license": True,
260
301
  "copyright": True,
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: fosslight_source
3
- Version: 2.3.9
3
+ Version: 2.3.11
4
4
  Summary: FOSSLight Source Scanner
5
5
  Author: LG Electronics
6
6
  License-Expression: Apache-2.0
@@ -20,7 +20,7 @@ Requires-Dist: setuptools<=80.10.2
20
20
  Requires-Dist: pyparsing
21
21
  Requires-Dist: scanoss>=1.45.0
22
22
  Requires-Dist: XlsxWriter
23
- Requires-Dist: fosslight_util>=2.2.8
23
+ Requires-Dist: fosslight_util>=2.2.12
24
24
  Requires-Dist: PyYAML
25
25
  Requires-Dist: wheel>=0.38.1
26
26
  Requires-Dist: intbitset
@@ -32,6 +32,7 @@ Requires-Dist: normality==2.6.1
32
32
  Requires-Dist: psycopg2-binary>=2.9.10; python_version >= "3.13"
33
33
  Requires-Dist: tomli; python_version < "3.11"
34
34
  Requires-Dist: tqdm
35
+ Requires-Dist: truststore
35
36
  Dynamic: license-file
36
37
 
37
38
  <!--
@@ -3,6 +3,7 @@ MANIFEST.in
3
3
  README.md
4
4
  pyproject.toml
5
5
  src/fosslight_source/__init__.py
6
+ src/fosslight_source/_exclude.py
6
7
  src/fosslight_source/_help.py
7
8
  src/fosslight_source/_kb_client.py
8
9
  src/fosslight_source/_license_matched.py
@@ -22,9 +23,11 @@ src/fosslight_source.egg-info/dependency_links.txt
22
23
  src/fosslight_source.egg-info/entry_points.txt
23
24
  src/fosslight_source.egg-info/requires.txt
24
25
  src/fosslight_source.egg-info/top_level.txt
26
+ tests/test_kb_ssl.py
25
27
  tests/test_manifest_android_bp.py
26
28
  tests/test_manifest_composer.py
27
29
  tests/test_manifest_pyproject.py
28
30
  tests/test_manifest_recommended_scenarios.py
31
+ tests/test_multi_value_order.py
29
32
  tests/test_parsing_unknown_spdx.py
30
33
  tests/test_tox.py
@@ -2,7 +2,7 @@ setuptools<=80.10.2
2
2
  pyparsing
3
3
  scanoss>=1.45.0
4
4
  XlsxWriter
5
- fosslight_util>=2.2.8
5
+ fosslight_util>=2.2.12
6
6
  PyYAML
7
7
  wheel>=0.38.1
8
8
  intbitset
@@ -11,6 +11,7 @@ lxml>=6.0.1
11
11
  fingerprints==1.2.3
12
12
  normality==2.6.1
13
13
  tqdm
14
+ truststore
14
15
 
15
16
  [:platform_system == "Darwin" and platform_machine == "x86_64"]
16
17
  cryptography<49
@@ -0,0 +1,52 @@
1
+ # Copyright (c) 2026 LG Electronics Inc.
2
+ # SPDX-License-Identifier: Apache-2.0
3
+ """Tests for KB HTTPS SSL context selection."""
4
+
5
+ import ssl
6
+ from unittest.mock import patch
7
+
8
+ from fosslight_source._kb_client import create_kb_ssl_context, kb_ssl_verify_enabled
9
+
10
+
11
+ def test_kb_ssl_verify_enabled_default(monkeypatch):
12
+ monkeypatch.delenv("KB_SSL_VERIFY", raising=False)
13
+ assert kb_ssl_verify_enabled() is True
14
+
15
+
16
+ def test_kb_ssl_verify_disabled_values(monkeypatch):
17
+ for value in ("false", "0", "no", "OFF"):
18
+ monkeypatch.setenv("KB_SSL_VERIFY", value)
19
+ assert kb_ssl_verify_enabled() is False
20
+
21
+
22
+ def test_create_kb_ssl_context_insecure(monkeypatch):
23
+ monkeypatch.setenv("KB_SSL_VERIFY", "false")
24
+ ctx = create_kb_ssl_context()
25
+ assert ctx.verify_mode == ssl.CERT_NONE
26
+ assert ctx.check_hostname is False
27
+
28
+
29
+ def test_create_kb_ssl_context_uses_truststore_when_verify_on(monkeypatch):
30
+ monkeypatch.setenv("KB_SSL_VERIFY", "true")
31
+ ctx = create_kb_ssl_context()
32
+ assert ctx.verify_mode != ssl.CERT_NONE
33
+ assert ctx.check_hostname is True
34
+
35
+
36
+ def test_kb_request_passes_ssl_context():
37
+ from fosslight_source._kb_client import _kb_request
38
+
39
+ class _Resp:
40
+ def read(self):
41
+ return b"{}"
42
+
43
+ def __enter__(self):
44
+ return self
45
+
46
+ def __exit__(self, *args):
47
+ return False
48
+
49
+ with patch("fosslight_source._kb_client.urllib.request.urlopen", return_value=_Resp()) as urlopen:
50
+ _kb_request("https://kb.example/", "health")
51
+ assert urlopen.call_args.kwargs["context"] is not None
52
+ assert isinstance(urlopen.call_args.kwargs["context"], ssl.SSLContext)
@@ -49,7 +49,7 @@ def test_merge_results_sets_manifest_flag_without_overwriting_scancode_licenses(
49
49
 
50
50
  assert len(merged) == 1
51
51
  assert merged[0].is_manifest_file is True
52
- assert merged[0].licenses == ["Apache-2.0", "MIT", "BSD"]
52
+ assert merged[0].licenses == ["Apache-2.0", "BSD", "MIT"]
53
53
 
54
54
 
55
55
  def test_merge_results_skips_android_bp_not_in_scancode_result():
@@ -18,7 +18,7 @@ def test_scenario1_android_bp_keeps_scancode_licenses():
18
18
 
19
19
  assert len(merged) == 1
20
20
  assert merged[0].is_manifest_file is True
21
- assert merged[0].licenses == ["Apache-2.0", "unknown-license-reference", "BSD", "MIT", "OFL"]
21
+ assert merged[0].licenses == ["Apache-2.0", "BSD", "MIT", "OFL", "unknown-license-reference"]
22
22
 
23
23
 
24
24
  def test_scenario2_package_json_manifest_fail_keeps_scancode_licenses():
@@ -0,0 +1,26 @@
1
+ # Copyright (c) 2026 LG Electronics Inc.
2
+ # SPDX-License-Identifier: Apache-2.0
3
+ """Stable order for multi-value License / Copyright cells."""
4
+
5
+ from fosslight_source._merge import _get_top_merge_values
6
+ from fosslight_source._scan_item import SourceItem
7
+
8
+
9
+ def test_licenses_are_stored_sorted():
10
+ item = SourceItem("dummy.c")
11
+ item.licenses = ["zlib", "mit", "apache-2.0"]
12
+ assert item.licenses == ["apache-2.0", "mit", "zlib"]
13
+
14
+
15
+ def test_top_merge_copyrights_break_count_ties_alphabetically():
16
+ items = []
17
+ for text in ["Copyright Z", "Copyright A", "Copyright Z", "Copyright M"]:
18
+ item = SourceItem("dummy.c")
19
+ item.copyright = [text]
20
+ items.append(item)
21
+
22
+ assert _get_top_merge_values(items, lambda i: i.copyright) == [
23
+ "Copyright Z",
24
+ "Copyright A",
25
+ "Copyright M",
26
+ ]
@@ -9,7 +9,9 @@ from fosslight_source._parsing_scancode_file_item import (
9
9
  _declared_licenses_from_matched_text,
10
10
  build_comment_from_detected_expression,
11
11
  parsing_scancode,
12
- _matched_texts_with_other_licenses,
12
+ _file_has_other_license,
13
+ _matched_text_has_http_url,
14
+ _should_suppress_unknown_license_reference,
13
15
  )
14
16
 
15
17
 
@@ -126,7 +128,7 @@ def test_unknown_spdx_comment_preserves_and_or_from_detected_expression():
126
128
  success, results, _messages, _ = parsing_scancode(scancode_file_list)
127
129
 
128
130
  assert success is True
129
- assert results[0].licenses == ["NEW", "DApache-2.0", "GPL-2.0"]
131
+ assert results[0].licenses == ["DApache-2.0", "GPL-2.0", "NEW"]
130
132
  assert "unknown-license-reference" not in [lic.lower() for lic in results[0].licenses]
131
133
  assert results[0].comment == "NEW OR DApache-2.0 AND GPL-2.0"
132
134
 
@@ -157,6 +159,168 @@ def test_unknown_license_reference_suppressed_when_same_matched_text_has_other_l
157
159
  assert results[0].licenses == ["GPL-2.0"]
158
160
 
159
161
 
162
+ @pytest.mark.parametrize(
163
+ "matched_text",
164
+ [
165
+ "License terms: http://example.com/license",
166
+ "License terms: https://example.com/license",
167
+ "License terms: HTTPS://example.com/license",
168
+ ],
169
+ )
170
+ def test_unknown_license_reference_kept_when_it_is_only_license_with_url(matched_text):
171
+ scancode_file_list = [{
172
+ "path": "reference.txt",
173
+ "type": "file",
174
+ "license_detections": [{
175
+ "matches": [{
176
+ "license_expression": "unknown-license-reference",
177
+ "matched_text": matched_text,
178
+ }],
179
+ }],
180
+ "copyrights": [],
181
+ }]
182
+
183
+ success, results, _messages, license_list = parsing_scancode(scancode_file_list)
184
+
185
+ assert success is True
186
+ assert results[0].licenses == ["unknown-license-reference"]
187
+ assert [item.matched_text for item in license_list.values()] == [matched_text]
188
+
189
+
190
+ @pytest.mark.parametrize("matched_text", ["See the accompanying license", "", None])
191
+ def test_unknown_license_reference_suppressed_without_url(matched_text):
192
+ scancode_file_list = [{
193
+ "path": "reference.txt",
194
+ "type": "file",
195
+ "license_detections": [{
196
+ "matches": [{
197
+ "license_expression": "unknown-license-reference",
198
+ "matched_text": matched_text,
199
+ }],
200
+ }],
201
+ "copyrights": [],
202
+ }]
203
+
204
+ success, results, _messages, license_list = parsing_scancode(scancode_file_list)
205
+
206
+ assert success is True
207
+ assert results[0].licenses == []
208
+ assert license_list == {}
209
+
210
+
211
+ def test_unknown_license_reference_suppressed_by_other_license_in_same_file():
212
+ reference_text = "License terms: https://example.com/license"
213
+ scancode_file_list = [{
214
+ "path": "mixed.txt",
215
+ "type": "file",
216
+ "detected_license_expression": (
217
+ "unknown-license-reference OR mit OR apache-2.0"
218
+ ),
219
+ "license_detections": [{
220
+ "matches": [
221
+ {
222
+ "license_expression": "unknown-license-reference",
223
+ "matched_text": reference_text,
224
+ },
225
+ {
226
+ "license_expression": "mit",
227
+ "matched_text": "Permission is hereby granted, free of charge...",
228
+ },
229
+ {
230
+ "license_expression": "apache-2.0",
231
+ "matched_text": "Licensed under the Apache License, Version 2.0",
232
+ },
233
+ ],
234
+ }],
235
+ "copyrights": [],
236
+ }]
237
+
238
+ success, results, _messages, license_list = parsing_scancode(scancode_file_list)
239
+
240
+ assert success is True
241
+ assert results[0].licenses == ["MIT", "Apache-2.0"]
242
+ assert results[0].comment == "MIT OR Apache-2.0"
243
+ assert all(
244
+ item.license != "unknown-license-reference"
245
+ for item in license_list.values()
246
+ )
247
+
248
+
249
+ def test_unknown_license_reference_filter_is_scoped_to_each_file():
250
+ reference_text = "License terms: https://example.com/license"
251
+ scancode_file_list = [
252
+ {
253
+ "path": "reference.txt",
254
+ "type": "file",
255
+ "license_detections": [{
256
+ "matches": [{
257
+ "license_expression": "unknown-license-reference",
258
+ "matched_text": reference_text,
259
+ }],
260
+ }],
261
+ "copyrights": [],
262
+ },
263
+ {
264
+ "path": "mit.txt",
265
+ "type": "file",
266
+ "license_detections": [{
267
+ "matches": [{
268
+ "license_expression": "mit",
269
+ "matched_text": "Permission is hereby granted, free of charge...",
270
+ }],
271
+ }],
272
+ "copyrights": [],
273
+ },
274
+ ]
275
+
276
+ success, results, _messages, _license_list = parsing_scancode(scancode_file_list)
277
+
278
+ assert success is True
279
+ assert results[0].licenses == ["unknown-license-reference"]
280
+ assert results[1].licenses == ["MIT"]
281
+
282
+
283
+ def test_only_url_backed_unknown_license_reference_is_reported():
284
+ valid_text = "License terms: https://example.com/license"
285
+ invalid_text = "See the accompanying license"
286
+ scancode_file_list = [{
287
+ "path": "references.txt",
288
+ "type": "file",
289
+ "license_detections": [{
290
+ "matches": [
291
+ {
292
+ "license_expression": "unknown-license-reference",
293
+ "matched_text": invalid_text,
294
+ },
295
+ {
296
+ "license_expression": "unknown-license-reference",
297
+ "matched_text": valid_text,
298
+ },
299
+ ],
300
+ }],
301
+ "copyrights": [],
302
+ }]
303
+
304
+ success, results, _messages, license_list = parsing_scancode(scancode_file_list)
305
+
306
+ assert success is True
307
+ assert results[0].licenses == ["unknown-license-reference"]
308
+ assert [item.matched_text for item in license_list.values()] == [valid_text]
309
+
310
+
311
+ @pytest.mark.parametrize(
312
+ ("matched_text", "expected"),
313
+ [
314
+ ("http://example.com/license", True),
315
+ ("HTTPS://example.com/license", True),
316
+ ("ftp://example.com/license", False),
317
+ (None, False),
318
+ ],
319
+ )
320
+ def test_matched_text_has_http_url(matched_text, expected):
321
+ assert _matched_text_has_http_url(matched_text) is expected
322
+
323
+
160
324
  def test_licenseref_tokens_stripped_from_unknown_spdx_and_expression():
161
325
  scancode_file_list = [{
162
326
  "path": "refs.py",
@@ -196,11 +360,11 @@ def test_build_comment_from_detected_expression_helper():
196
360
  "matched_text": matched_same,
197
361
  },
198
362
  ]
199
- other_texts = _matched_texts_with_other_licenses(matches)
363
+ has_other_license = _file_has_other_license(matches)
200
364
  comment = build_comment_from_detected_expression(
201
365
  "(unknown-spdx OR unknown-spdx) AND unknown-license-reference AND gpl-2.0",
202
366
  matches,
203
- other_texts,
367
+ _should_suppress_unknown_license_reference(matches, has_other_license),
204
368
  )
205
369
  assert comment == "NEW OR DApache-2.0 AND GPL-2.0"
206
370
 
@@ -216,7 +380,7 @@ def test_comment_without_parens_uses_operator_before_kept_token():
216
380
  comment = build_comment_from_detected_expression(
217
381
  "mit OR unknown-license-reference AND apache-2.0",
218
382
  matches,
219
- {matched_same},
383
+ True,
220
384
  )
221
385
  assert comment == "MIT AND Apache-2.0"
222
386
 
@@ -232,7 +396,7 @@ def test_comment_with_parens_preserves_or_group():
232
396
  comment = build_comment_from_detected_expression(
233
397
  "mit OR (unknown-license-reference AND apache-2.0)",
234
398
  matches,
235
- {matched_same},
399
+ True,
236
400
  )
237
401
  assert comment == "MIT OR Apache-2.0"
238
402
 
@@ -248,7 +412,7 @@ def test_comment_with_parens_preserves_and_after_group():
248
412
  comment = build_comment_from_detected_expression(
249
413
  "(mit OR unknown-license-reference) AND apache-2.0",
250
414
  matches,
251
- {matched_same},
415
+ True,
252
416
  )
253
417
  assert comment == "MIT AND Apache-2.0"
254
418
 
@@ -337,10 +501,10 @@ def test_android_bp_soong_license_kinds_without_line_comment_in_license():
337
501
  licenses = results[0].licenses
338
502
  assert licenses == [
339
503
  "Apache-2.0",
340
- "unknown-license-reference",
341
504
  "BSD",
342
505
  "MIT",
343
506
  "OFL",
507
+ "unknown-license-reference",
344
508
  ]
345
509
  assert all("//" not in lic for lic in results[0].licenses)
346
510
  assert all('"' not in lic for lic in results[0].licenses)