# SPDX-FileCopyrightText: 2024-2025 SPDX contributors
# SPDX-FileType: SOURCE
# SPDX-License-Identifier: Apache-2.0
"""Base checking functionality."""
from __future__ import annotations
import logging
import os
import warnings
from abc import ABC, abstractmethod
from typing import TYPE_CHECKING, Any, cast
from spdx_tools.spdx.parser import parse_anything
from spdx_tools.spdx.parser.error import SPDXParsingError
from spdx_tools.spdx.validation.document_validator import validate_full_spdx_document
from .adapters import NullAdapter, SbomAdapter, Spdx2Adapter, Spdx3Adapter
from .constants import DEFAULT_SBOM_SPEC
from .graph_utils import analyze_graph_connectivity
from .report import (
ReportContext,
report_html,
report_json,
report_text,
)
from .spdx3_utils import (
parse_spdx3_file,
validate_spdx3_data,
)
if TYPE_CHECKING:
from collections.abc import Sized
from spdx_python_model.bindings import v3_0_1 as spdx3
from spdx_tools.spdx.model.document import Document
from spdx_tools.spdx.validation.validation_message import ValidationMessage
# pylint: disable=too-many-instance-attributes,too-many-public-methods
[docs]
class BaseChecker(ABC):
"""Base class for all compliance/conformance checkers.
This base class contains methods for common tasks like file parsing
and information extractions from the SBOM.
Any class inheriting from BaseChecker must implement its abstract methods,
such as `check_compliance` and `output_json`.
"""
# Minimum elements/baseline attributes required by a compliance standard
MIN_ELEMENTS: list[str] = []
# Mapping of components without information
# SBOM component name: (list containing components missing the info, label)
_COMPONENTS_WITHOUT_INFO = {
"name": ("components_without_names", "Components missing a name"),
"version": ("components_without_versions", "Components missing a version"),
"identifier": (
"components_without_identifiers",
"Components missing an identifier",
),
"supplier": ("components_without_suppliers", "Components missing a supplier"),
"concluded_license": (
"components_without_concluded_licenses",
"Components missing a concluded license",
),
"copyright_text": (
"components_without_copyright_texts",
"Components missing a copyright text",
),
}
compliance_standard: str = "" # fsct3-min, ntia
sbom_spec: str = "" # spdx2, spdx3
# These are detectable by spdx-tools, so not needed for now.
# file_format: str = "" # json, rdf-xml, tag-value, yaml, xml
file: str = ""
# For SPDX 3, we have to use SHACLObjectSet instead of SpdxDocument,
# because we need access to relationships and other elements that are not
# accessible from SpdxDocument.
doc: Document | spdx3.SHACLObjectSet | None = None
__spdx3_doc: spdx3.SpdxDocument | None = None # cached SPDX 3 document
_parsing_errors: list[str] = []
_validation_messages: list[ValidationMessage] = []
_conformance_messages: list[ValidationMessage] = []
sbom_name: str = ""
# Lists of components missing required information.
# Each item is a tuple of (component name, component SPDX ID).
components_without_names: list[tuple[str, str]]
components_without_versions: list[tuple[str, str]]
components_without_suppliers: list[tuple[str, str]]
components_without_identifiers: list[tuple[str, str]]
components_without_concluded_licenses: list[tuple[str, str]]
components_without_copyright_texts: list[tuple[str, str]]
# (info name, components missing that info) pairs.
all_components_without_info: list[tuple[str, list[tuple[str, str]]]]
sbom_gen_context: list[str] # SBOM types (SPDX 3 only)
doc_version: bool = False # Has SPDX document version?
doc_author: bool = False # Has SPDX document author?
doc_timestamp: bool = False # Has SPDX document creation timestamp?
dependency_relationships: bool = False # Has dependency relationship?
# See https://github.com/spdx/ntia-conformance-checker/issues/392
# for discussion on dependency relationships and DESCRIBES.
# False when no component is reachable from the SBOM root.
components_evaluated: bool = False
compliant: bool = False # Is SBOM compliant with the chosen standard?
@property
def ntia_minimum_elements_compliant(self) -> bool:
"""Deprecated: use ``compliant`` instead."""
warnings.warn(
"ntia_minimum_elements_compliant is deprecated; use compliant instead.",
DeprecationWarning,
stacklevel=2,
)
return self.compliant
@property
def parsing_errors(self) -> list[str]:
"""Parsing errors encountered during file parsing."""
return self._parsing_errors
@property
def parsing_error(self) -> list[str]:
"""Deprecated: use ``parsing_errors`` instead."""
warnings.warn(
"parsing_error is deprecated; use parsing_errors instead.",
DeprecationWarning,
stacklevel=2,
)
return self._parsing_errors
@property
def validation_messages(self) -> list[ValidationMessage]:
"""Validation messages from SPDX document validation."""
return self._validation_messages
@property
def conformance_messages(self) -> list[ValidationMessage]:
"""Conformance messages from compliance/conformance checks."""
return self._conformance_messages
[docs]
@abstractmethod
def check_compliance(self) -> bool:
"""Abstract method to check compliance/conformance."""
raise NotImplementedError
def __init__(
self,
file: str,
validate: bool = True,
compliance: str = "",
sbom_spec: str = DEFAULT_SBOM_SPEC,
) -> None:
"""
Initialize the BaseChecker.
Args:
file (str): The name of the file to be checked.
validate (bool): Whether to validate the file.
compliance (str): The compliance standard to be used.
sbom_spec (str): The SBOM specification to be used.
"""
self.compliance_standard = compliance
self.sbom_spec = sbom_spec
# self.file_format = ""
self.file = file
# Make sure the logs are instance variables and not class variables
# to avoid shared state between instances.
self._parsing_errors = []
self._validation_messages = []
self._conformance_messages = []
self._init_result_lists()
self.reachable_component_ids: set[str] = set()
self.floating_component_ids: set[str] = set()
self.unknown_pointer_edges: dict[str, list[str]] = {}
# "Pointers" refers to relationship edges targeting unknown/missing elements in the graph.
self.has_unknown_pointers: bool = False
self.adapter: SbomAdapter = NullAdapter()
match sbom_spec:
case "spdx2":
self.doc = self.parse_file()
case "spdx3":
object_set, parsing_errors = parse_spdx3_file(self.file)
self._parsing_errors.extend(parsing_errors)
if not object_set:
logging.error("Failed to parse the SPDX 3 file.")
else:
self.doc = object_set
self.__spdx3_doc, _val_msgs = validate_spdx3_data(object_set)
if not self.__spdx3_doc or _val_msgs:
logging.error("SpdxDocument not found or invalid.")
self._validation_messages.extend(_val_msgs)
case _:
# We can add a heuristic to detect the spec from the file content here,
# in case sbom_spec is not provided or invalid.
raise ValueError(f"Unsupported SBOM specification: {sbom_spec}")
if self.doc:
if self.sbom_spec == "spdx2":
self.doc = cast("Document", self.doc)
self.adapter = Spdx2Adapter(self.doc)
elif self.sbom_spec == "spdx3":
self.adapter = Spdx3Adapter(
cast("spdx3.SHACLObjectSet", self.doc),
self.__spdx3_doc,
)
self._evaluate_graph_connectivity()
if validate and sbom_spec == "spdx2":
self.doc = cast("Document", self.doc)
self._validation_messages = validate_full_spdx_document(self.doc)
self.sbom_name = self.get_sbom_name()
self.sbom_gen_context = self.get_sbom_types()
self.doc_version = self.check_doc_version()
self.doc_author = self.check_author()
self.doc_timestamp = self.check_timestamp()
self.dependency_relationships = self.check_dependency_relationships()
self.components_without_names = self.get_components_without_names()
self.components_without_versions = self.get_components_without_versions()
self.components_without_suppliers = self.get_components_without_suppliers()
self.components_without_identifiers = (
self.get_components_without_identifiers()
)
self.components_without_concluded_licenses = (
self.get_components_without_concluded_licenses()
)
self.components_without_copyright_texts = (
self.get_components_without_copyright_texts()
)
self.all_components_without_info = self._get_all_components_without_info()
self.table_elements: list[tuple[str, bool]] = []
def _init_result_lists(self) -> None:
"""
Initialize per-instance result lists.
Class-level defaults would be shared between instances whenever
parsing fails, so each instance gets its own empty lists.
"""
self.components_without_names = []
self.components_without_versions = []
self.components_without_suppliers = []
self.components_without_identifiers = []
self.components_without_concluded_licenses = []
self.components_without_copyright_texts = []
self.all_components_without_info = []
self.sbom_gen_context = []
def _all_provided(self, missing: Sized) -> bool:
"""
Check if components were evaluated and none is missing the information.
Args:
missing: Components missing the information.
Returns:
bool: True if components were evaluated and ``missing`` is empty.
"""
return self.components_evaluated and not missing
[docs]
def check_doc_version(self) -> bool:
"""Check if the document's specification version exists."""
return self.adapter.check_doc_version()
[docs]
def check_author(self) -> bool:
"""Check if the author of SBOM data exists."""
return self.adapter.check_author()
[docs]
def check_dependency_relationships(self) -> bool:
"""Check if the SBOM document declares dependency information."""
return self.adapter.check_dependency_relationships()
[docs]
def check_timestamp(self) -> bool:
"""Check if the SBOM creation timestamp exists."""
return self.adapter.check_timestamp()
[docs]
def get_doc_spec_version(self) -> str | None:
"""Retrieve the document's specification version."""
return self.adapter.get_doc_spec_version()
[docs]
def get_sbom_name(self) -> str:
"""Retrieve the name of the SBOM."""
return self.adapter.get_sbom_name()
[docs]
def get_sbom_types(self) -> list[str]:
"""Get SBOM types (generation context) from the document.
CISA Framing Software Component Transparency (2024) listed
"SBOM type" as one of baseline attributes, see Table 1 (p. 22) in:
https://www.cisa.gov/resources-tools/resources/framing-software-component-transparency-2024
"""
# SBOM type is only available in SPDX 3
return self.adapter.get_sbom_types(self._conformance_messages)
[docs]
def get_components_without_concluded_licenses(self) -> list[tuple[str, str]]:
"""
Retrieve components missing a concluded license.
Returns:
list[tuple[str, str]]: A list of tuples of the form
(component_name, spdx_id). Consumers should extract the
preferred value (name or SPDX ID) as needed.
"""
# Note: concluded license is mandatory in SPDX-2.2 and SPDX-2.3
return self.adapter.get_components_without_concluded_licenses(
self.reachable_component_ids
)
[docs]
def get_components_without_copyright_texts(self) -> list[tuple[str, str]]:
"""
Retrieve components missing a copyright text.
Returns:
list[tuple[str, str]]: A list of tuples of the form
(component_name, spdx_id). Consumers should extract the
preferred value (name or SPDX ID) as needed.
"""
return self.adapter.get_components_without_copyright_texts(
self.reachable_component_ids
)
[docs]
def get_components_without_identifiers(self) -> list[tuple[str, str]]:
"""
Retrieve components missing unique identifiers (SPDX IDs).
Returns:
list[tuple[str, str]]: A list of tuples of the form
(component_name, spdx_id). Consumers should extract the
preferred value (name or SPDX ID) as needed.
"""
return self.adapter.get_components_without_identifiers(
self.reachable_component_ids
)
[docs]
def get_components_without_names(self) -> list[tuple[str, str]]:
"""
Retrieve components missing a name.
Returns:
list[tuple[str, str]]: A list of tuples of the form
(component_name, spdx_id). Consumers should extract the
preferred value (name or SPDX ID) as needed.
"""
return self.adapter.get_components_without_names(self.reachable_component_ids)
[docs]
def get_components_without_suppliers(self) -> list[tuple[str, str]]:
"""
Retrieve components missing supplier information.
Returns:
list[tuple[str, str]]: A list of tuples of the form
(component_name, spdx_id). Consumers should extract the
preferred value (name or SPDX ID) as needed.
"""
return self.adapter.get_components_without_suppliers(
self.reachable_component_ids
)
[docs]
def get_components_without_versions(self) -> list[tuple[str, str]]:
"""
Retrieve components missing version information.
Returns:
list[tuple[str, str]]: A list of tuples of the form
(component_name, spdx_id). Consumers should extract the
preferred value (name or SPDX ID) as needed.
"""
return self.adapter.get_components_without_versions(
self.reachable_component_ids
)
def _get_all_components_without_info(
self,
) -> list[tuple[str, list[tuple[str, str]]]]:
"""Get a list of components missing information for each required info."""
# If all lists are empty, return an empty list
if all(
not getattr(self, list_name, [])
for list_name, _ in self._COMPONENTS_WITHOUT_INFO.values()
):
return []
return [
(info_name, getattr(self, self._COMPONENTS_WITHOUT_INFO[info_name][0], []))
for info_name in self.MIN_ELEMENTS
if info_name in self._COMPONENTS_WITHOUT_INFO
and getattr(self, self._COMPONENTS_WITHOUT_INFO[info_name][0], [])
]
[docs]
def get_total_number_components(self) -> int:
"""
Retrieve total number of components.
Returns:
int: The total number of components.
"""
return self.adapter.get_total_number_components()
[docs]
def parse_file(self) -> Document | None:
"""
Parse SPDX 2 SBOM document.
Returns:
Document | None: An SPDX 2 SBOM document if successful, otherwise None.
"""
if not self.file or str(self.file).strip() == "":
logging.error("No file path provided.")
return None
if not os.path.exists(self.file):
logging.error("File not found: %s", self.file)
return None
try:
# Annotate as `object` to avoid differences between local and CI
# mypy stubs for `parse_anything.parse_file`. Casting to
# `Document` below makes the return type explicit for callers.
doc: object = parse_anything.parse_file(self.file)
except SPDXParsingError as err:
# err.get_messages() is untyped in spdx-tools; cast to Any to
# silence mypy's "no-untyped-call" check in typed contexts.
self._parsing_errors.extend(cast("Any", err).get_messages())
return None
except Exception as err: # pylint: disable=broad-except
# Catch any other errors, including BeartypeCallHintParamViolation
# from the spdx-tools library when parsing invalid SPDX files.
# The spdx-tools library uses beartype for runtime type checking,
# which throws exceptions when encountering missing required fields
# (e.g., missing author, timestamp, or identifiers).
logging.debug("Error parsing file: %s", err)
self._parsing_errors.append(
f"Error parsing file: {type(err).__name__}: {str(err)}"
)
return None
return cast("Document", doc)
[docs]
def print_components_missing_info(self) -> None:
"""
Print information about components that are missing required details.
What is considered "missing" is determined by a compliance standard.
Subclasses may override this method to provide custom behavior.
Returns:
None
"""
# If parsing failed, skip
if self._parsing_errors:
return
if not self.all_components_without_info:
return
print("Missing required information in these components:")
for info_name, components in self.all_components_without_info:
print(
f"{info_name} ({len(components)}): "
f"{', '.join([name for name, _ in components])}"
)
[docs]
def print_table_output(self, verbose: bool = False) -> None:
"""
Print element-by-element result table.
Args:
verbose (bool): If True, print detailed information.
Returns:
None
"""
report_context = ReportContext(
sbom_spec=getattr(self, "sbom_spec", ""),
compliance_standard=getattr(self, "compliance_standard", ""),
compliant=getattr(self, "compliant", False),
requirement_results=getattr(self, "table_elements", []),
components_without_info=getattr(self, "all_components_without_info", []),
validation_messages=self._validation_messages,
conformance_messages=self._conformance_messages,
parsing_errors=self._parsing_errors,
unknown_pointer_edges=getattr(self, "unknown_pointer_edges", {}),
floating_component_ids=getattr(self, "floating_component_ids", set()),
components_evaluated=self.components_evaluated,
)
print(report_text(report_context, verbose))
[docs]
def output_html(self) -> str:
"""
Create element-by-element result table in HTML.
Returns:
str: The HTML representation of the results.
"""
report_context = ReportContext(
sbom_spec=getattr(self, "sbom_spec", ""),
compliance_standard=getattr(self, "compliance_standard", ""),
compliant=getattr(self, "compliant", False),
requirement_results=getattr(self, "table_elements", []),
components_without_info=getattr(self, "all_components_without_info", []),
validation_messages=self._validation_messages,
conformance_messages=self._conformance_messages,
parsing_errors=self._parsing_errors,
unknown_pointer_edges=getattr(self, "unknown_pointer_edges", {}),
floating_component_ids=getattr(self, "floating_component_ids", set()),
components_evaluated=self.components_evaluated,
)
return report_html(report_context, verbose=True)
[docs]
def output_json(self) -> dict[str, Any]:
"""
Create a JSON-serializable result dict.
Subclasses may override to provide custom fields.
"""
return report_json(self)
def _evaluate_graph_connectivity(self) -> None:
"""Evaluate graph connectivity to isolate floating nodes and unknown pointers."""
reachable, floating, unknown_pointer_edges, has_unknown_pointers = (
analyze_graph_connectivity(
self.sbom_spec, self.doc, getattr(self, "_BaseChecker__spdx3_doc", None)
)
)
self.reachable_component_ids = reachable
self.components_evaluated = bool(reachable)
self.floating_component_ids = floating
self.unknown_pointer_edges = unknown_pointer_edges
self.has_unknown_pointers = has_unknown_pointers
if self.floating_component_ids:
logging.warning(
"Found %d disconnected 'floating' elements. They will be ignored for compliance.",
len(self.floating_component_ids),
)
if self.has_unknown_pointers:
logging.error(
"Unknown components detected!"
" A relationship points to a missing element."
)