diff --git a/src/fosslight_source/cli.py b/src/fosslight_source/cli.py index b34c40f..6fa625f 100755 --- a/src/fosslight_source/cli.py +++ b/src/fosslight_source/cli.py @@ -572,7 +572,7 @@ def run_scanners( excluded_path_without_dot, excluded_files, cnt_file_except_skipped) = get_excluded_paths(path_to_scan, path_to_exclude_with_filename) - logger.debug(f"Skipped paths: {excluded_path_with_default_exclusion}") + logger.debug(f"Skipped paths count: {len(excluded_path_with_default_exclusion)}") if not selected_scanner: selected_scanner = ALL_MODE @@ -580,8 +580,8 @@ def run_scanners( success, result_log[RESULT_KEY], scancode_result, license_list = run_scan( path_to_scan, output_file_name, write_json_file, num_cores, True, print_matched_text, formats, called_by_cli, time_out, correct_mode, - correct_filepath, excluded_path_with_default_exclusion, - excluded_files, hide_progress, + correct_filepath, path_to_exclude, + hide_progress=hide_progress, ) excluded_files = set(excluded_files) if excluded_files else set() if selected_scanner in ['scanoss', ALL_MODE]: diff --git a/src/fosslight_source/run_scancode.py b/src/fosslight_source/run_scancode.py index 8c24a01..5c2b99e 100755 --- a/src/fosslight_source/run_scancode.py +++ b/src/fosslight_source/run_scancode.py @@ -22,7 +22,7 @@ PACKAGE_DIRECTORY, ) from commoncode.fileset import is_included -from typing import Tuple, Iterable +from typing import Callable, Tuple logger = logging.getLogger(constant.LOGGER_NAME) warnings.filterwarnings("ignore", category=FutureWarning) @@ -62,6 +62,96 @@ def _apply_scancode_unset_workaround(kwargs: dict) -> None: logger.debug("scancode UNSET workaround skipped: %s", ex) +_WILDCARD_EXTENSIONS = { + "png", "mp3", "wav", "comp", "bin", "o", "db", "tflite", + "ttf", "exe", "dll", "jpg", "jpeg", "gif", + "zip", "tar", "tgz", "gz", + "bmp", "webp", "ico", +} | {ext.lower() for ext in EXCLUDE_FILE_EXTENSION} +_SKIP_DIR_NAMES = frozenset(name.lower() for name in PACKAGE_DIRECTORY + EXCLUDE_DIRECTORY) +_SKIP_EXTS = frozenset(ext.lower() for ext in EXCLUDE_FILE_EXTENSION) + + +def _normalize_custom_pattern(pattern: str, abs_path_to_scan: str) -> set: + pat = pattern.replace('\\', '/').strip() + if not pat: + return set() + + patterns_to_add = {pat} + + if pat.endswith("/**"): + base = pat[:-3].rstrip("/") + if base: + patterns_to_add.add(base) + elif pat.endswith("/*"): + base = pat[:-2].rstrip("/") + if base: + patterns_to_add.add(base) + patterns_to_add.add(f"{base}/**") + elif pat.endswith("/"): + base = pat.rstrip("/") + if base: + patterns_to_add.add(base) + patterns_to_add.add(f"{base}/**") + patterns_to_add.add(f"{base}/*") + else: + full_path = os.path.join(abs_path_to_scan, pat) + if os.path.isdir(full_path): + patterns_to_add.add(f"{pat}/**") + patterns_to_add.add(f"{pat}/*") + + return patterns_to_add + + +def _expand_custom_exclude_pattern(pattern: str, abs_path_to_scan: str) -> set: + exclude_path_normalized = os.path.normpath( + pattern.replace('\\', '/').strip() + ).replace("\\", "/") + if not exclude_path_normalized: + return set() + + patterns = set(_normalize_custom_pattern(exclude_path_normalized, abs_path_to_scan)) + + if exclude_path_normalized.endswith("/**"): + base_dir = exclude_path_normalized[:-3].rstrip("/") + if base_dir: + full_exclude_path = os.path.join(abs_path_to_scan, base_dir) + if os.path.isdir(full_exclude_path): + patterns.add(base_dir) + patterns.add(exclude_path_normalized) + else: + patterns.add(exclude_path_normalized) + return patterns + + has_glob_chars = any(char in exclude_path_normalized for char in ['*', '?', '[']) + if has_glob_chars: + patterns.add(exclude_path_normalized) + return patterns + + full_exclude_path = os.path.join(abs_path_to_scan, exclude_path_normalized) + if os.path.isdir(full_exclude_path): + base_path = exclude_path_normalized.rstrip("/") + if base_path: + patterns.add(base_path) + patterns.add(f"{base_path}/**") + else: + patterns.add(exclude_path_normalized) + elif os.path.isfile(full_exclude_path): + ext = os.path.splitext(exclude_path_normalized)[1].lstrip('.').lower() + if ext in _WILDCARD_EXTENSIONS: + patterns.add(f"*.{ext}") + else: + patterns.add(f"**/{exclude_path_normalized}") + else: + ext = os.path.splitext(exclude_path_normalized)[1].lstrip('.').lower() + if ext in _WILDCARD_EXTENSIONS: + patterns.add(f"*.{ext}") + else: + patterns.add(exclude_path_normalized) + + return patterns + + def _directory_ignore_pattern(dir_name: str) -> str: """Path-based glob for a directory name (avoids matching the scan root itself).""" normalized = dir_name.strip().strip("/").replace("\\", "/") @@ -70,7 +160,10 @@ def _directory_ignore_pattern(dir_name: str) -> str: return f"**/{normalized}/**" -def _default_scancode_coarse_ignore_patterns() -> frozenset: +def _default_scancode_coarse_ignore_patterns( + path_to_exclude: list = None, + abs_path_to_scan: str = "" +) -> frozenset: """ Coarse ignore patterns aligned with fosslight_util.get_excluded_paths() rules. Directory names use path-based globs (e.g. **/tests/**) so they do not match @@ -83,73 +176,93 @@ def _default_scancode_coarse_ignore_patterns() -> frozenset: patterns.add(f"*.{ext}") for name in EXCLUDE_FILENAME: patterns.add(name) + + for pattern in path_to_exclude or []: + if os.path.isabs(pattern): + pattern = os.path.relpath(pattern, abs_path_to_scan) + patterns.update(_expand_custom_exclude_pattern(pattern, abs_path_to_scan)) + return frozenset(patterns) -def _is_covered_by_coarse_ignore(rel_path: str, coarse_patterns: Iterable[str]) -> bool: - excludes = {pattern: "" for pattern in coarse_patterns} +def _to_excludes_dict(patterns) -> dict: + return {pattern: "exclude" for pattern in patterns} + + +def _is_path_covered(rel_path: str, excludes: dict) -> bool: return not is_included(rel_path, includes={}, excludes=excludes) -def _add_path_to_exclude_pattern( +def _add_ignore_pattern( patterns: set, - exclude_path: str, - abs_path_to_scan: str, + excludes: dict, + pattern: str, + *, + sample_path: str = None, +) -> bool: + if pattern in patterns: + return False + if sample_path and _is_path_covered(sample_path, excludes): + return False + patterns.add(pattern) + excludes[pattern] = "exclude" + return True + + +def _make_pre_scan_skip_filter( coarse_patterns: frozenset, +) -> Tuple[dict, Callable[[str, str], bool], Callable[[str, str], bool]]: + excludes = _to_excludes_dict(coarse_patterns) + + def should_skip_dir(dir_name: str, rel_dir: str) -> bool: + if dir_name.startswith('.'): + return True + if dir_name.lower() in _SKIP_DIR_NAMES: + return True + return _is_path_covered(f"{rel_dir}/_", excludes) + + def should_skip_file(file_name: str, rel_path: str) -> bool: + if file_name.startswith('.'): + return True + ext = os.path.splitext(file_name)[1].lstrip('.').lower() + if ext in _SKIP_EXTS: + return True + return _is_path_covered(rel_path, excludes) + + return excludes, should_skip_dir, should_skip_file + + +def _add_binary_ignore_patterns( + patterns: set, + excludes: dict, + binary_paths: list, ) -> None: - exclude_path_normalized = os.path.normpath(exclude_path).replace("\\", "/") + extensions = set() + no_ext_paths = [] - if exclude_path_normalized.endswith("/**"): - base_dir = exclude_path_normalized[:-3].rstrip("/") - if base_dir: - full_exclude_path = os.path.join(abs_path_to_scan, base_dir) - if os.path.isdir(full_exclude_path): - patterns.add(base_dir) - patterns.add(exclude_path_normalized) - else: - patterns.add(exclude_path_normalized) + for rel_path in binary_paths: + ext = os.path.splitext(rel_path)[1].lstrip('.').lower() + if ext: + extensions.add(ext) else: - patterns.add(exclude_path_normalized) - return + no_ext_paths.append(rel_path) - has_glob_chars = any(char in exclude_path_normalized for char in ['*', '?', '[']) - if has_glob_chars: - patterns.add(exclude_path_normalized) - return + for ext in extensions: + _add_ignore_pattern(patterns, excludes, f"*.{ext}") - full_exclude_path = os.path.join(abs_path_to_scan, exclude_path_normalized) - if os.path.isdir(full_exclude_path): - base_path = exclude_path_normalized.rstrip("/") - if base_path: - patterns.add(base_path) - patterns.add(f"{base_path}/**") - else: - patterns.add(exclude_path_normalized) - elif os.path.isfile(full_exclude_path): - if not _is_covered_by_coarse_ignore(exclude_path_normalized, coarse_patterns): - patterns.add(f"**/{exclude_path_normalized}") - else: - patterns.add(exclude_path_normalized) + for rel_path in no_ext_paths: + _add_ignore_pattern( + patterns, excludes, f"**/{rel_path}", sample_path=rel_path + ) def _build_scancode_ignore_patterns( - path_to_exclude: list, - abs_path_to_scan: str, + coarse_patterns: frozenset, binary_paths: list, ) -> tuple: - coarse_patterns = _default_scancode_coarse_ignore_patterns() patterns = set(coarse_patterns) - - for path in path_to_exclude or []: - if os.path.isabs(path): - exclude_path = os.path.relpath(path, abs_path_to_scan) - else: - exclude_path = path - _add_path_to_exclude_pattern(patterns, exclude_path, abs_path_to_scan, coarse_patterns) - - for rel_path in binary_paths: - patterns.add(f"**/{rel_path}") - + excludes = _to_excludes_dict(coarse_patterns) + _add_binary_ignore_patterns(patterns, excludes, binary_paths) return tuple(sorted(patterns)) @@ -198,8 +311,13 @@ def run_scan( output_json_file = "" if not called_by_cli: - logger, _result_log = init_log(os.path.join(output_path, f"fosslight_log_src_{_file_time}.txt"), - True, logging.INFO, logging.DEBUG, _PKG_NAME, path_to_scan, path_to_exclude) + log_file_path = os.path.join( + output_path, f"fosslight_log_src_{_file_time}.txt" + ) + logger, _result_log = init_log( + log_file_path, True, logging.INFO, logging.DEBUG, + _PKG_NAME, path_to_scan, path_to_exclude + ) logger.info(f"Tool Info : {_result_log['Tool Info']}") @@ -214,21 +332,36 @@ def run_scan( pretty_params["output_file"] = output_file_name abs_path_to_scan = os.path.abspath(path_to_scan) binary_paths = [] - for root, _, files in os.walk(path_to_scan): + coarse_patterns = _default_scancode_coarse_ignore_patterns( + path_to_exclude, abs_path_to_scan + ) + _, should_skip_dir, should_skip_file = _make_pre_scan_skip_filter( + coarse_patterns + ) + + for root, dirs, files in os.walk(path_to_scan): + rel_root = os.path.relpath(root, abs_path_to_scan).replace("\\", "/") + dirs[:] = [ + d for d in dirs + if not should_skip_dir( + d, d if rel_root == "." else f"{rel_root}/{d}" + ) + ] for name in files: + rel_path = name if rel_root == "." else f"{rel_root}/{name}" + if should_skip_file(name, rel_path): + continue full_path = os.path.join(root, name) try: if not check_binary(full_path, True): continue except Exception: continue - rel_path = os.path.relpath(full_path, abs_path_to_scan) - rel_norm = os.path.normpath(rel_path).replace("\\", "/") - binary_paths.append(rel_norm) - logger.debug(f"Excluded binary from scancode: {rel_norm}") + binary_paths.append(rel_path) + logger.debug(f"Excluded binary from scancode: {rel_path}") ignore_tuple = _build_scancode_ignore_patterns( - path_to_exclude, abs_path_to_scan, binary_paths + coarse_patterns, binary_paths ) logger.debug(f"Scancode ignore patterns: {len(ignore_tuple)}")