Source code for ntia_conformance_checker.base_checker

# SPDX-FileCopyrightText: 2024-2025 SPDX contributors
# SPDX-FileType: SOURCE
# SPDX-License-Identifier: Apache-2.0

"""Base checking functionality."""

from __future__ import annotations

import logging
import os
import warnings
from abc import ABC, abstractmethod
from typing import TYPE_CHECKING, Any, cast

from spdx_tools.spdx.parser import parse_anything
from spdx_tools.spdx.parser.error import SPDXParsingError
from spdx_tools.spdx.validation.document_validator import validate_full_spdx_document

from .adapters import NullAdapter, SbomAdapter, Spdx2Adapter, Spdx3Adapter
from .constants import DEFAULT_SBOM_SPEC
from .graph_utils import analyze_graph_connectivity
from .report import (
    ReportContext,
    report_html,
    report_json,
    report_text,
)
from .spdx3_utils import (
    parse_spdx3_file,
    validate_spdx3_data,
)

if TYPE_CHECKING:
    from collections.abc import Sized

    from spdx_python_model.bindings import v3_0_1 as spdx3
    from spdx_tools.spdx.model.document import Document
    from spdx_tools.spdx.validation.validation_message import ValidationMessage


# pylint: disable=too-many-instance-attributes,too-many-public-methods
[docs] class BaseChecker(ABC): """Base class for all compliance/conformance checkers. This base class contains methods for common tasks like file parsing and information extractions from the SBOM. Any class inheriting from BaseChecker must implement its abstract methods, such as `check_compliance` and `output_json`. """ # Minimum elements/baseline attributes required by a compliance standard MIN_ELEMENTS: list[str] = [] # Mapping of components without information # SBOM component name: (list containing components missing the info, label) _COMPONENTS_WITHOUT_INFO = { "name": ("components_without_names", "Components missing a name"), "version": ("components_without_versions", "Components missing a version"), "identifier": ( "components_without_identifiers", "Components missing an identifier", ), "supplier": ("components_without_suppliers", "Components missing a supplier"), "concluded_license": ( "components_without_concluded_licenses", "Components missing a concluded license", ), "copyright_text": ( "components_without_copyright_texts", "Components missing a copyright text", ), } compliance_standard: str = "" # fsct3-min, ntia sbom_spec: str = "" # spdx2, spdx3 # These are detectable by spdx-tools, so not needed for now. # file_format: str = "" # json, rdf-xml, tag-value, yaml, xml file: str = "" # For SPDX 3, we have to use SHACLObjectSet instead of SpdxDocument, # because we need access to relationships and other elements that are not # accessible from SpdxDocument. doc: Document | spdx3.SHACLObjectSet | None = None __spdx3_doc: spdx3.SpdxDocument | None = None # cached SPDX 3 document _parsing_errors: list[str] = [] _validation_messages: list[ValidationMessage] = [] _conformance_messages: list[ValidationMessage] = [] sbom_name: str = "" # Lists of components missing required information. # Each item is a tuple of (component name, component SPDX ID). components_without_names: list[tuple[str, str]] components_without_versions: list[tuple[str, str]] components_without_suppliers: list[tuple[str, str]] components_without_identifiers: list[tuple[str, str]] components_without_concluded_licenses: list[tuple[str, str]] components_without_copyright_texts: list[tuple[str, str]] # (info name, components missing that info) pairs. all_components_without_info: list[tuple[str, list[tuple[str, str]]]] sbom_gen_context: list[str] # SBOM types (SPDX 3 only) doc_version: bool = False # Has SPDX document version? doc_author: bool = False # Has SPDX document author? doc_timestamp: bool = False # Has SPDX document creation timestamp? dependency_relationships: bool = False # Has dependency relationship? # See https://github.com/spdx/ntia-conformance-checker/issues/392 # for discussion on dependency relationships and DESCRIBES. # False when no component is reachable from the SBOM root. components_evaluated: bool = False compliant: bool = False # Is SBOM compliant with the chosen standard? @property def ntia_minimum_elements_compliant(self) -> bool: """Deprecated: use ``compliant`` instead.""" warnings.warn( "ntia_minimum_elements_compliant is deprecated; use compliant instead.", DeprecationWarning, stacklevel=2, ) return self.compliant @property def parsing_errors(self) -> list[str]: """Parsing errors encountered during file parsing.""" return self._parsing_errors @property def parsing_error(self) -> list[str]: """Deprecated: use ``parsing_errors`` instead.""" warnings.warn( "parsing_error is deprecated; use parsing_errors instead.", DeprecationWarning, stacklevel=2, ) return self._parsing_errors @property def validation_messages(self) -> list[ValidationMessage]: """Validation messages from SPDX document validation.""" return self._validation_messages @property def conformance_messages(self) -> list[ValidationMessage]: """Conformance messages from compliance/conformance checks.""" return self._conformance_messages
[docs] @abstractmethod def check_compliance(self) -> bool: """Abstract method to check compliance/conformance.""" raise NotImplementedError
def __init__( self, file: str, validate: bool = True, compliance: str = "", sbom_spec: str = DEFAULT_SBOM_SPEC, ) -> None: """ Initialize the BaseChecker. Args: file (str): The name of the file to be checked. validate (bool): Whether to validate the file. compliance (str): The compliance standard to be used. sbom_spec (str): The SBOM specification to be used. """ self.compliance_standard = compliance self.sbom_spec = sbom_spec # self.file_format = "" self.file = file # Make sure the logs are instance variables and not class variables # to avoid shared state between instances. self._parsing_errors = [] self._validation_messages = [] self._conformance_messages = [] self._init_result_lists() self.reachable_component_ids: set[str] = set() self.floating_component_ids: set[str] = set() self.unknown_pointer_edges: dict[str, list[str]] = {} # "Pointers" refers to relationship edges targeting unknown/missing elements in the graph. self.has_unknown_pointers: bool = False self.adapter: SbomAdapter = NullAdapter() match sbom_spec: case "spdx2": self.doc = self.parse_file() case "spdx3": object_set, parsing_errors = parse_spdx3_file(self.file) self._parsing_errors.extend(parsing_errors) if not object_set: logging.error("Failed to parse the SPDX 3 file.") else: self.doc = object_set self.__spdx3_doc, _val_msgs = validate_spdx3_data(object_set) if not self.__spdx3_doc or _val_msgs: logging.error("SpdxDocument not found or invalid.") self._validation_messages.extend(_val_msgs) case _: # We can add a heuristic to detect the spec from the file content here, # in case sbom_spec is not provided or invalid. raise ValueError(f"Unsupported SBOM specification: {sbom_spec}") if self.doc: if self.sbom_spec == "spdx2": self.doc = cast("Document", self.doc) self.adapter = Spdx2Adapter(self.doc) elif self.sbom_spec == "spdx3": self.adapter = Spdx3Adapter( cast("spdx3.SHACLObjectSet", self.doc), self.__spdx3_doc, ) self._evaluate_graph_connectivity() if validate and sbom_spec == "spdx2": self.doc = cast("Document", self.doc) self._validation_messages = validate_full_spdx_document(self.doc) self.sbom_name = self.get_sbom_name() self.sbom_gen_context = self.get_sbom_types() self.doc_version = self.check_doc_version() self.doc_author = self.check_author() self.doc_timestamp = self.check_timestamp() self.dependency_relationships = self.check_dependency_relationships() self.components_without_names = self.get_components_without_names() self.components_without_versions = self.get_components_without_versions() self.components_without_suppliers = self.get_components_without_suppliers() self.components_without_identifiers = ( self.get_components_without_identifiers() ) self.components_without_concluded_licenses = ( self.get_components_without_concluded_licenses() ) self.components_without_copyright_texts = ( self.get_components_without_copyright_texts() ) self.all_components_without_info = self._get_all_components_without_info() self.table_elements: list[tuple[str, bool]] = [] def _init_result_lists(self) -> None: """ Initialize per-instance result lists. Class-level defaults would be shared between instances whenever parsing fails, so each instance gets its own empty lists. """ self.components_without_names = [] self.components_without_versions = [] self.components_without_suppliers = [] self.components_without_identifiers = [] self.components_without_concluded_licenses = [] self.components_without_copyright_texts = [] self.all_components_without_info = [] self.sbom_gen_context = [] def _all_provided(self, missing: Sized) -> bool: """ Check if components were evaluated and none is missing the information. Args: missing: Components missing the information. Returns: bool: True if components were evaluated and ``missing`` is empty. """ return self.components_evaluated and not missing
[docs] def check_doc_version(self) -> bool: """Check if the document's specification version exists.""" return self.adapter.check_doc_version()
[docs] def check_author(self) -> bool: """Check if the author of SBOM data exists.""" return self.adapter.check_author()
[docs] def check_dependency_relationships(self) -> bool: """Check if the SBOM document declares dependency information.""" return self.adapter.check_dependency_relationships()
[docs] def check_timestamp(self) -> bool: """Check if the SBOM creation timestamp exists.""" return self.adapter.check_timestamp()
[docs] def get_doc_spec_version(self) -> str | None: """Retrieve the document's specification version.""" return self.adapter.get_doc_spec_version()
[docs] def get_sbom_name(self) -> str: """Retrieve the name of the SBOM.""" return self.adapter.get_sbom_name()
[docs] def get_sbom_types(self) -> list[str]: """Get SBOM types (generation context) from the document. CISA Framing Software Component Transparency (2024) listed "SBOM type" as one of baseline attributes, see Table 1 (p. 22) in: https://www.cisa.gov/resources-tools/resources/framing-software-component-transparency-2024 """ # SBOM type is only available in SPDX 3 return self.adapter.get_sbom_types(self._conformance_messages)
[docs] def get_components_without_concluded_licenses(self) -> list[tuple[str, str]]: """ Retrieve components missing a concluded license. Returns: list[tuple[str, str]]: A list of tuples of the form (component_name, spdx_id). Consumers should extract the preferred value (name or SPDX ID) as needed. """ # Note: concluded license is mandatory in SPDX-2.2 and SPDX-2.3 return self.adapter.get_components_without_concluded_licenses( self.reachable_component_ids )
[docs] def get_components_without_identifiers(self) -> list[tuple[str, str]]: """ Retrieve components missing unique identifiers (SPDX IDs). Returns: list[tuple[str, str]]: A list of tuples of the form (component_name, spdx_id). Consumers should extract the preferred value (name or SPDX ID) as needed. """ return self.adapter.get_components_without_identifiers( self.reachable_component_ids )
[docs] def get_components_without_names(self) -> list[tuple[str, str]]: """ Retrieve components missing a name. Returns: list[tuple[str, str]]: A list of tuples of the form (component_name, spdx_id). Consumers should extract the preferred value (name or SPDX ID) as needed. """ return self.adapter.get_components_without_names(self.reachable_component_ids)
[docs] def get_components_without_suppliers(self) -> list[tuple[str, str]]: """ Retrieve components missing supplier information. Returns: list[tuple[str, str]]: A list of tuples of the form (component_name, spdx_id). Consumers should extract the preferred value (name or SPDX ID) as needed. """ return self.adapter.get_components_without_suppliers( self.reachable_component_ids )
[docs] def get_components_without_versions(self) -> list[tuple[str, str]]: """ Retrieve components missing version information. Returns: list[tuple[str, str]]: A list of tuples of the form (component_name, spdx_id). Consumers should extract the preferred value (name or SPDX ID) as needed. """ return self.adapter.get_components_without_versions( self.reachable_component_ids )
def _get_all_components_without_info( self, ) -> list[tuple[str, list[tuple[str, str]]]]: """Get a list of components missing information for each required info.""" # If all lists are empty, return an empty list if all( not getattr(self, list_name, []) for list_name, _ in self._COMPONENTS_WITHOUT_INFO.values() ): return [] return [ (info_name, getattr(self, self._COMPONENTS_WITHOUT_INFO[info_name][0], [])) for info_name in self.MIN_ELEMENTS if info_name in self._COMPONENTS_WITHOUT_INFO and getattr(self, self._COMPONENTS_WITHOUT_INFO[info_name][0], []) ]
[docs] def get_total_number_components(self) -> int: """ Retrieve total number of components. Returns: int: The total number of components. """ return self.adapter.get_total_number_components()
[docs] def parse_file(self) -> Document | None: """ Parse SPDX 2 SBOM document. Returns: Document | None: An SPDX 2 SBOM document if successful, otherwise None. """ if not self.file or str(self.file).strip() == "": logging.error("No file path provided.") return None if not os.path.exists(self.file): logging.error("File not found: %s", self.file) return None try: # Annotate as `object` to avoid differences between local and CI # mypy stubs for `parse_anything.parse_file`. Casting to # `Document` below makes the return type explicit for callers. doc: object = parse_anything.parse_file(self.file) except SPDXParsingError as err: # err.get_messages() is untyped in spdx-tools; cast to Any to # silence mypy's "no-untyped-call" check in typed contexts. self._parsing_errors.extend(cast("Any", err).get_messages()) return None except Exception as err: # pylint: disable=broad-except # Catch any other errors, including BeartypeCallHintParamViolation # from the spdx-tools library when parsing invalid SPDX files. # The spdx-tools library uses beartype for runtime type checking, # which throws exceptions when encountering missing required fields # (e.g., missing author, timestamp, or identifiers). logging.debug("Error parsing file: %s", err) self._parsing_errors.append( f"Error parsing file: {type(err).__name__}: {str(err)}" ) return None return cast("Document", doc)
[docs] def print_components_missing_info(self) -> None: """ Print information about components that are missing required details. What is considered "missing" is determined by a compliance standard. Subclasses may override this method to provide custom behavior. Returns: None """ # If parsing failed, skip if self._parsing_errors: return if not self.all_components_without_info: return print("Missing required information in these components:") for info_name, components in self.all_components_without_info: print( f"{info_name} ({len(components)}): " f"{', '.join([name for name, _ in components])}" )
[docs] def print_table_output(self, verbose: bool = False) -> None: """ Print element-by-element result table. Args: verbose (bool): If True, print detailed information. Returns: None """ report_context = ReportContext( sbom_spec=getattr(self, "sbom_spec", ""), compliance_standard=getattr(self, "compliance_standard", ""), compliant=getattr(self, "compliant", False), requirement_results=getattr(self, "table_elements", []), components_without_info=getattr(self, "all_components_without_info", []), validation_messages=self._validation_messages, conformance_messages=self._conformance_messages, parsing_errors=self._parsing_errors, unknown_pointer_edges=getattr(self, "unknown_pointer_edges", {}), floating_component_ids=getattr(self, "floating_component_ids", set()), components_evaluated=self.components_evaluated, ) print(report_text(report_context, verbose))
[docs] def output_html(self) -> str: """ Create element-by-element result table in HTML. Returns: str: The HTML representation of the results. """ report_context = ReportContext( sbom_spec=getattr(self, "sbom_spec", ""), compliance_standard=getattr(self, "compliance_standard", ""), compliant=getattr(self, "compliant", False), requirement_results=getattr(self, "table_elements", []), components_without_info=getattr(self, "all_components_without_info", []), validation_messages=self._validation_messages, conformance_messages=self._conformance_messages, parsing_errors=self._parsing_errors, unknown_pointer_edges=getattr(self, "unknown_pointer_edges", {}), floating_component_ids=getattr(self, "floating_component_ids", set()), components_evaluated=self.components_evaluated, ) return report_html(report_context, verbose=True)
[docs] def output_json(self) -> dict[str, Any]: """ Create a JSON-serializable result dict. Subclasses may override to provide custom fields. """ return report_json(self)
def _evaluate_graph_connectivity(self) -> None: """Evaluate graph connectivity to isolate floating nodes and unknown pointers.""" reachable, floating, unknown_pointer_edges, has_unknown_pointers = ( analyze_graph_connectivity( self.sbom_spec, self.doc, getattr(self, "_BaseChecker__spdx3_doc", None) ) ) self.reachable_component_ids = reachable self.components_evaluated = bool(reachable) self.floating_component_ids = floating self.unknown_pointer_edges = unknown_pointer_edges self.has_unknown_pointers = has_unknown_pointers if self.floating_component_ids: logging.warning( "Found %d disconnected 'floating' elements. They will be ignored for compliance.", len(self.floating_component_ids), ) if self.has_unknown_pointers: logging.error( "Unknown components detected!" " A relationship points to a missing element." )