From 0b0d8b867034d45056632b2f1d71d19072ce68d7 Mon Sep 17 00:00:00 2001 From: Soim Kim Date: Thu, 10 Sep 2026 15:36:32 +0900 Subject: [PATCH 1/2] fix(scancode): omit Detected licenses already in License When SPDX priority drops body matches into a Detected comment, skip expressions that are already present in the License column to avoid redundant output such as Apache-2.0 in both fields. --- .../_parsing_scancode_file_item.py | 44 ++++++++++-- tests/test_parsing_unknown_spdx.py | 67 +++++++++++++++++++ 2 files changed, 105 insertions(+), 6 deletions(-) diff --git a/src/fosslight_source/_parsing_scancode_file_item.py b/src/fosslight_source/_parsing_scancode_file_item.py index 926da05..b2262a4 100755 --- a/src/fosslight_source/_parsing_scancode_file_item.py +++ b/src/fosslight_source/_parsing_scancode_file_item.py @@ -446,12 +446,44 @@ def _dedupe_detected_comment_expressions(expressions: list[str]) -> list[str]: return kept -def build_detected_comment_from_dropped_matches(dropped_matches: list) -> str: - """Build ``Detected: a, b OR c`` from matches dropped by SPDX priority.""" - displays = [ - _display_expression_for_detected_comment(match) - for match in (dropped_matches or []) +def _expression_covered_by_reported_licenses( + expression: str, reported_licenses: list +) -> bool: + """ + True when expression is already represented in License column values. + + Matches the whole expression, or every AND/OR token, case-insensitively. + """ + reported = { + (lic or "").strip().lower() + for lic in (reported_licenses or []) + if (lic or "").strip() + } + if not expression or not reported: + return False + expr = expression.strip() + if expr.lower() in reported: + return True + tokens = [ + token.strip().lower() + for token in split_spdx_expression(expr) + if token and token.strip() ] + return bool(tokens) and all(token in reported for token in tokens) + + +def build_detected_comment_from_dropped_matches( + dropped_matches: list, reported_licenses: list | None = None +) -> str: + """Build ``Detected: a, b OR c`` from matches dropped by SPDX priority.""" + displays = [] + for match in (dropped_matches or []): + display = _display_expression_for_detected_comment(match) + if not display: + continue + if _expression_covered_by_reported_licenses(display, reported_licenses or []): + continue + displays.append(display) deduped = _dedupe_detected_comment_expressions(displays) if not deduped: return "" @@ -840,7 +872,7 @@ def parsing_scancode( match for match in all_matches if id(match) not in process_ids ] detected_comment = build_detected_comment_from_dropped_matches( - dropped_matches + dropped_matches, license_detected ) if detected_comment: result_item.comment = detected_comment diff --git a/tests/test_parsing_unknown_spdx.py b/tests/test_parsing_unknown_spdx.py index 925b7a2..506c5ee 100644 --- a/tests/test_parsing_unknown_spdx.py +++ b/tests/test_parsing_unknown_spdx.py @@ -703,6 +703,73 @@ def test_dedupe_detected_comment_examples(dropped_exprs, expected_comment): assert build_detected_comment_from_dropped_matches(dropped) == expected_comment +def test_detected_comment_excludes_licenses_already_in_license_column(): + from fosslight_source._parsing_scancode_file_item import ( + build_detected_comment_from_dropped_matches, + ) + dropped = [ + { + "license_expression": "apache-2.0", + "license_expression_spdx": "Apache-2.0", + "matched_text": "Licensed under the Apache License, Version 2.0", + }, + { + "license_expression": "mit", + "license_expression_spdx": "MIT", + "matched_text": "Permission is hereby granted", + }, + ] + assert ( + build_detected_comment_from_dropped_matches(dropped, ["Apache-2.0"]) + == "Detected: MIT" + ) + assert build_detected_comment_from_dropped_matches(dropped, ["Apache-2.0", "MIT"]) == "" + + +def test_readme_spdx_apache_does_not_repeat_in_detected_comment(): + """Body apache-2.0 match dropped by SPDX priority must not echo License column.""" + scancode_file_list = [{ + "path": "README.md", + "type": "file", + "detected_license_expression": "apache-2.0", + "detected_license_expression_spdx": "Apache-2.0", + "license_detections": [ + { + "matches": [{ + "license_expression": "apache-2.0", + "license_expression_spdx": "Apache-2.0", + "matched_text": "SPDX-License-Identifier: Apache-2.0", + }], + }, + { + "matches": [ + { + "license_expression": "apache-2.0", + "license_expression_spdx": "Apache-2.0", + "matched_text": ( + 'Licensed under the Apache License, Version 2.0 ' + '(the "License"); you may not use this file except ' + "in compliance with the License." + ), + }, + { + "license_expression": "apache-2.0", + "license_expression_spdx": "Apache-2.0", + "matched_text": "SPDX-License-Identifier: Apache-2.0", + }, + ], + }, + ], + "copyrights": [], + }] + + success, results, _messages, _ = parsing_scancode(scancode_file_list) + + assert success is True + assert results[0].licenses == ["Apache-2.0"] + assert results[0].comment == "" + + def test_spdx_priority_detected_comment_uses_spdx_ids(): """Dropped body matches use license_expression_spdx in Detected comment.""" scancode_file_list = [{ From 53168356416ef10bebb17fd73dc005bf52525fca Mon Sep 17 00:00:00 2001 From: Soim Kim Date: Thu, 10 Sep 2026 15:51:29 +0900 Subject: [PATCH 2/2] fix(scancode): filter Detected comment by final License values Use result_item.licenses after the SourceItem setter so licenses removed by length limits are not wrongly suppressed from the Detected comment. --- .../_parsing_scancode_file_item.py | 2 +- tests/test_parsing_unknown_spdx.py | 37 +++++++++++++++++++ 2 files changed, 38 insertions(+), 1 deletion(-) diff --git a/src/fosslight_source/_parsing_scancode_file_item.py b/src/fosslight_source/_parsing_scancode_file_item.py index b2262a4..5aa0a93 100755 --- a/src/fosslight_source/_parsing_scancode_file_item.py +++ b/src/fosslight_source/_parsing_scancode_file_item.py @@ -872,7 +872,7 @@ def parsing_scancode( match for match in all_matches if id(match) not in process_ids ] detected_comment = build_detected_comment_from_dropped_matches( - dropped_matches, license_detected + dropped_matches, result_item.licenses ) if detected_comment: result_item.comment = detected_comment diff --git a/tests/test_parsing_unknown_spdx.py b/tests/test_parsing_unknown_spdx.py index 506c5ee..0b8f55f 100644 --- a/tests/test_parsing_unknown_spdx.py +++ b/tests/test_parsing_unknown_spdx.py @@ -726,6 +726,43 @@ def test_detected_comment_excludes_licenses_already_in_license_column(): assert build_detected_comment_from_dropped_matches(dropped, ["Apache-2.0", "MIT"]) == "" +def test_detected_comment_uses_final_licenses_after_column_limit(): + """License dropped by column length limit must still be allowed in Detected.""" + fillers = [f"AaaLic{i:02d}-{'y' * 80}" for i in range(20)] + spdx_matches = [ + { + "license_expression": name.lower(), + "license_expression_spdx": name, + "matched_text": f"SPDX-License-Identifier: {name}", + } + for name in fillers + ["ZebraExtra"] + ] + scancode_file_list = [{ + "path": "crowded.c", + "type": "file", + "detected_license_expression": " AND ".join( + name.lower() for name in fillers + ["ZebraExtra"] + ), + "license_detections": [ + {"matches": spdx_matches}, + { + "matches": [{ + "license_expression": "zebraextra", + "license_expression_spdx": "ZebraExtra", + "matched_text": "This file is under ZebraExtra terms.", + }], + }, + ], + "copyrights": [], + }] + + success, results, _messages, _ = parsing_scancode(scancode_file_list) + + assert success is True + assert "ZebraExtra" not in results[0].licenses + assert "Detected: ZebraExtra" in results[0].comment + + def test_readme_spdx_apache_does_not_repeat_in_detected_comment(): """Body apache-2.0 match dropped by SPDX priority must not echo License column.""" scancode_file_list = [{