aboutsummaryrefslogtreecommitdiffhomepage
path: root/sec_certs/sample
diff options
context:
space:
mode:
authoradamjanovsky2022-12-09 17:10:19 +0100
committerGitHub2022-12-09 17:10:19 +0100
commit73b3b0c361f9545450fa188bec50606d64bb1afd (patch)
tree0a1f9034c309ba88e5f72a31634b014c23a57df5 /sec_certs/sample
parent19338dc9fd9ab257c36cfa277994abe202e97de2 (diff)
downloadsec-certs-73b3b0c361f9545450fa188bec50606d64bb1afd.tar.gz
sec-certs-73b3b0c361f9545450fa188bec50606d64bb1afd.tar.zst
sec-certs-73b3b0c361f9545450fa188bec50606d64bb1afd.zip
flat -> src layout (#294)
- Some mypy fixes - Flat layout -> src layout - Ditch `setup.py` and `setup.cfg` in favour of `pyproject.toml` - Non-pinned requirements moved from `requirements/*.in` to `pyproject.toml`
Diffstat (limited to 'sec_certs/sample')
-rw-r--r--sec_certs/sample/__init__.py33
-rw-r--r--sec_certs/sample/cc_certificate_id.py151
-rw-r--r--sec_certs/sample/cc_maintenance_update.py99
-rw-r--r--sec_certs/sample/certificate.py90
-rw-r--r--sec_certs/sample/common_criteria.py986
-rw-r--r--sec_certs/sample/cpe.py100
-rw-r--r--sec_certs/sample/cve.py185
-rw-r--r--sec_certs/sample/fips.py649
-rw-r--r--sec_certs/sample/fips_algorithm.py50
-rw-r--r--sec_certs/sample/fips_iut.py166
-rw-r--r--sec_certs/sample/fips_mip.py242
-rw-r--r--sec_certs/sample/protection_profile.py55
-rw-r--r--sec_certs/sample/sar.py59
13 files changed, 0 insertions, 2865 deletions
diff --git a/sec_certs/sample/__init__.py b/sec_certs/sample/__init__.py
deleted file mode 100644
index ecbbd541..00000000
--- a/sec_certs/sample/__init__.py
+++ /dev/null
@@ -1,33 +0,0 @@
-"""This package holds mostly data objects of primary interest (Common Criteria, FIPS), or assisting objects
-like CPE, CVE, etc. The objects mostly hold data and allow for serialization, but can also perform some basic transformations.
-"""
-
-from sec_certs.sample.cc_certificate_id import CertificateId
-from sec_certs.sample.cc_maintenance_update import CommonCriteriaMaintenanceUpdate
-from sec_certs.sample.common_criteria import CommonCriteriaCert
-from sec_certs.sample.cpe import CPE, cached_cpe
-from sec_certs.sample.cve import CVE
-from sec_certs.sample.fips import FIPSCertificate
-from sec_certs.sample.fips_algorithm import FIPSAlgorithm
-from sec_certs.sample.fips_iut import IUTEntry, IUTSnapshot
-from sec_certs.sample.fips_mip import MIPEntry, MIPSnapshot, MIPStatus
-from sec_certs.sample.protection_profile import ProtectionProfile
-from sec_certs.sample.sar import SAR
-
-__all__ = [
- "CertificateId",
- "CommonCriteriaMaintenanceUpdate",
- "CommonCriteriaCert",
- "CPE",
- "cached_cpe",
- "CVE",
- "FIPSCertificate",
- "FIPSAlgorithm",
- "IUTEntry",
- "IUTSnapshot",
- "MIPEntry",
- "MIPSnapshot",
- "MIPStatus",
- "ProtectionProfile",
- "SAR",
-]
diff --git a/sec_certs/sample/cc_certificate_id.py b/sec_certs/sample/cc_certificate_id.py
deleted file mode 100644
index 428ca73b..00000000
--- a/sec_certs/sample/cc_certificate_id.py
+++ /dev/null
@@ -1,151 +0,0 @@
-from __future__ import annotations
-
-import re
-from dataclasses import dataclass
-
-
-@dataclass(eq=True, frozen=True)
-class CertificateId:
- """
- A Common Criteria certificate id.
- """
-
- scheme: str
- raw: str
-
- def _canonical_fr(self) -> str:
- new_cert_id = self.clean
- rules = [
- "(?:Rapport de certification|Certification Report) ([0-9]+[/-_][0-9]+(?:v[1-9])?(?:[_/-][MSR][0-9]+)?)",
- "(?:ANSS[Ii]|DCSSI)(?:-CC)?[- ]([0-9]+[/-_][0-9]+(?:v[1-9])?(?:[_/-][MSR][0-9]+)?)",
- "([0-9]+[/-_][0-9]+(?:v[1-9])?(?:[_/-][MSR][0-9]+)?)",
- ]
- for rule in rules:
- if match := re.match(rule, new_cert_id):
- return "ANSSI-CC-" + match.group(1).replace("_", "/")
-
- return new_cert_id
-
- def _canonical_de(self) -> str:
- def extract_parts(bsi_parts: list[str]) -> tuple:
- cert_num = None
- cert_version = None
- cert_year = None
-
- if len(bsi_parts) > 3:
- cert_num = bsi_parts[3]
- if len(bsi_parts) > 4:
- if bsi_parts[4].startswith("V") or bsi_parts[4].startswith("v"):
- cert_version = bsi_parts[4].upper() # get version in uppercase
- else:
- cert_year = bsi_parts[4]
- if len(bsi_parts) > 5:
- cert_year = bsi_parts[5]
-
- return cert_num, cert_version, cert_year
-
- bsi_parts = self.clean.split("-")
-
- cert_num, cert_version, cert_year = extract_parts(bsi_parts)
-
- # reconstruct BSI number again
- new_cert_id = "BSI-DSZ-CC"
- if cert_num is not None:
- new_cert_id += "-" + cert_num
- if cert_version is not None:
- new_cert_id += "-" + cert_version
- if cert_year is not None:
- new_cert_id += "-" + cert_year
-
- return new_cert_id
-
- def _canonical_es(self) -> str:
- cert_id = self.clean
- spain_parts = cert_id.split("-")
- cert_year = spain_parts[0]
- cert_batch = spain_parts[1].lstrip("0")
- cert_num = spain_parts[3].lstrip("0")
-
- if "v" in cert_num:
- cert_num = cert_num[: cert_num.find("v")]
- if "V" in cert_num:
- cert_num = cert_num[: cert_num.find("V")]
-
- new_cert_id = f"{cert_year}-{cert_batch}-INF-{cert_num.strip()}" # drop version # TODO: Maybe do not drop?
-
- return new_cert_id
-
- def _canonical_it(self):
- new_cert_id = self.clean
- if not new_cert_id.endswith("/RC"):
- new_cert_id = new_cert_id + "/RC"
-
- return new_cert_id
-
- def _canonical_in(self):
- return self.clean.replace(" ", "")
-
- def _canonical_se(self):
- return self.clean.replace(" ", "")
-
- def _canonical_uk(self):
- new_cert_id = self.clean
- if match := re.match("CERTIFICATION REPORT No. P([0-9]+[A-Z]?)", new_cert_id):
- new_cert_id = "CRP" + match.group(1)
- return new_cert_id
-
- def _canonical_ca(self):
- new_cert_id = self.clean
- if new_cert_id.endswith("-CR"):
- new_cert_id = new_cert_id[:-3]
- if new_cert_id.endswith("P"):
- new_cert_id = new_cert_id[:-1]
- return new_cert_id.replace(" ", "-")
-
- def _canonical_jp(self):
- new_cert_id = self.clean
- if match := re.match("Certification No. (C[0-9]+)", new_cert_id):
- return match.group(1)
- if match := re.search("CRP-(C[0-9]+)-", new_cert_id):
- return match.group(1)
- return new_cert_id
-
- def _canonical_no(self):
- new_cert_id = self.clean
- cert_num = int(new_cert_id.split("-")[1])
- return f"SERTIT-{cert_num:03}"
-
- @property
- def clean(self) -> str:
- """
- The clean version of this certificate id.
- """
- return self.raw.replace("\N{HYPHEN}", "-").strip()
-
- @property
- def canonical(self) -> str:
- """
- The canonical version of this certificate id.
- """
- # We have rules for some schemes to make canonical cert_ids.
- schemes = {
- "FR": self._canonical_fr,
- "DE": self._canonical_de,
- "ES": self._canonical_es,
- "IT": self._canonical_it,
- "IN": self._canonical_in,
- "SE": self._canonical_se,
- "UK": self._canonical_uk,
- "CA": self._canonical_ca,
- "JP": self._canonical_jp,
- "NO": self._canonical_no,
- }
-
- if self.scheme in schemes:
- return schemes[self.scheme]()
- else:
- return self.clean
-
-
-def canonicalize(cert_id_str: str, scheme: str) -> str:
- return CertificateId(scheme, cert_id_str).canonical
diff --git a/sec_certs/sample/cc_maintenance_update.py b/sec_certs/sample/cc_maintenance_update.py
deleted file mode 100644
index abe3b176..00000000
--- a/sec_certs/sample/cc_maintenance_update.py
+++ /dev/null
@@ -1,99 +0,0 @@
-from __future__ import annotations
-
-import logging
-from datetime import date
-from typing import ClassVar
-
-import sec_certs.utils.helpers as helpers
-from sec_certs.sample.common_criteria import CommonCriteriaCert
-from sec_certs.serialization.json import ComplexSerializableType
-
-logger = logging.getLogger(__name__)
-
-
-class CommonCriteriaMaintenanceUpdate(CommonCriteriaCert, ComplexSerializableType):
- pandas_columns: ClassVar[list[str]] = [
- "dgst",
- "name",
- "report_link",
- "st_link",
- "related_cert_digest",
- "maintenance_date",
- ]
-
- def __init__(
- self,
- name: str,
- report_link: str,
- st_link: str,
- state: CommonCriteriaCert.InternalState | None,
- pdf_data: CommonCriteriaCert.PdfData | None,
- heuristics: CommonCriteriaCert.Heuristics | None,
- related_cert_digest: str,
- maintenance_date: date,
- ):
- super().__init__(
- "",
- "",
- name,
- "",
- "",
- "",
- None,
- None,
- report_link,
- st_link,
- "",
- "",
- set(),
- set(),
- state,
- pdf_data,
- heuristics,
- )
- self.related_cert_digest = related_cert_digest
- self.maintenance_date = maintenance_date
-
- @property
- def serialized_attributes(self) -> list[str]:
- return ["dgst"] + list(self.__class__.__init__.__code__.co_varnames)[1:]
-
- @property
- def dgst(self) -> str:
- if not self.name:
- raise RuntimeError("MaintenanceUpdate digest can't be computed, because name of update is missing.")
- return "cert_" + self.related_cert_digest + "_update_" + helpers.get_first_16_bytes_sha256(self.name)
-
- @property
- def pandas_tuple(self) -> tuple:
- return tuple([getattr(self, x) for x in CommonCriteriaMaintenanceUpdate.pandas_columns])
-
- @classmethod
- def from_dict(cls, dct: dict) -> CommonCriteriaMaintenanceUpdate:
- dct.pop("dgst")
- return cls(*(tuple(dct.values())))
-
- @classmethod
- def get_updates_from_cc_cert(cls, cert: CommonCriteriaCert) -> list[CommonCriteriaMaintenanceUpdate]:
- if cert.maintenance_updates is None:
- raise RuntimeError("Dataset was probably not built correctly - this should not be happening.")
-
- return [
- cls(
- x.maintenance_title,
- x.maintenance_report_link,
- x.maintenance_st_link,
- None,
- None,
- None,
- cert.dgst,
- x.maintenance_date,
- )
- for x in cert.maintenance_updates
- if (
- x.maintenance_title is not None
- and x.maintenance_report_link is not None
- and x.maintenance_st_link is not None
- and x.maintenance_date is not None
- )
- ]
diff --git a/sec_certs/sample/certificate.py b/sec_certs/sample/certificate.py
deleted file mode 100644
index bb49c0df..00000000
--- a/sec_certs/sample/certificate.py
+++ /dev/null
@@ -1,90 +0,0 @@
-from __future__ import annotations
-
-import copy
-import logging
-from abc import ABC, abstractmethod
-from collections import ChainMap
-from dataclasses import dataclass, field
-from typing import Any, Generic, TypeVar
-
-import sec_certs.utils.extract
-from sec_certs.cert_rules import PANDAS_KEYWORDS_CATEGORIES
-from sec_certs.serialization.json import ComplexSerializableType
-
-logger = logging.getLogger(__name__)
-
-T = TypeVar("T", bound="Certificate")
-H = TypeVar("H", bound="Heuristics")
-P = TypeVar("P", bound="PdfData")
-
-
-@dataclass
-class References(ComplexSerializableType):
- directly_referenced_by: set[str] | None = field(default=None)
- indirectly_referenced_by: set[str] | None = field(default=None)
- directly_referencing: set[str] | None = field(default=None)
- indirectly_referencing: set[str] | None = field(default=None)
-
-
-class Heuristics:
- cpe_matches: set[str] | None
- related_cves: set[str] | None
-
-
-class PdfData:
- def get_keywords_df_data(self, var: str) -> dict[str, float]:
- data_dct = getattr(self, var)
- return dict(
- ChainMap(
- *[
- sec_certs.utils.extract.get_sums_for_rules_subset(data_dct, cat)
- for cat in PANDAS_KEYWORDS_CATEGORIES
- ]
- )
- )
-
-
-class Certificate(Generic[T, H, P], ABC, ComplexSerializableType):
- manufacturer: str | None
- name: str | None
- pdf_data: P
- heuristics: H
-
- def __init__(self, *args, **kwargs):
- pass
-
- def __repr__(self) -> str:
- return str(self.to_dict())
-
- def __str__(self) -> str:
- return "Not implemented"
-
- @property
- @abstractmethod
- def dgst(self):
- raise NotImplementedError("Not meant to be implemented")
-
- @property
- @abstractmethod
- def label_studio_title(self):
- raise NotImplementedError("Not meant to be implemented")
-
- def __eq__(self, other: object) -> bool:
- if not isinstance(other, Certificate):
- return False
- return self.dgst == other.dgst
-
- def to_dict(self) -> dict[str, Any]:
- return {
- **{"dgst": self.dgst},
- **{key: val for key, val in copy.deepcopy(self.__dict__).items() if key in self.serialized_attributes},
- }
-
- @classmethod
- def from_dict(cls: type[T], dct: dict) -> T:
- dct.pop("dgst")
- return cls(**dct)
-
- @abstractmethod
- def compute_heuristics_version(self) -> None:
- raise NotImplementedError("Not meant to be implemented")
diff --git a/sec_certs/sample/common_criteria.py b/sec_certs/sample/common_criteria.py
deleted file mode 100644
index acd54178..00000000
--- a/sec_certs/sample/common_criteria.py
+++ /dev/null
@@ -1,986 +0,0 @@
-from __future__ import annotations
-
-import copy
-import re
-from collections import Counter, defaultdict
-from dataclasses import dataclass, field
-from datetime import date, datetime
-from enum import Enum
-from pathlib import Path
-from typing import Any, ClassVar
-from urllib.parse import unquote_plus, urlparse
-
-import numpy as np
-import requests
-from bs4 import Tag
-
-import sec_certs.utils.extract
-import sec_certs.utils.pdf
-import sec_certs.utils.sanitization
-from sec_certs import constants as constants
-from sec_certs.cert_rules import SARS_IMPLIED_FROM_EAL, cc_rules, rules, security_level_csv_scan
-from sec_certs.sample.cc_certificate_id import canonicalize
-from sec_certs.sample.certificate import Certificate
-from sec_certs.sample.certificate import Heuristics as BaseHeuristics
-from sec_certs.sample.certificate import PdfData as BasePdfData
-from sec_certs.sample.certificate import References, logger
-from sec_certs.sample.protection_profile import ProtectionProfile
-from sec_certs.sample.sar import SAR
-from sec_certs.serialization.json import ComplexSerializableType
-from sec_certs.serialization.pandas import PandasSerializableType
-from sec_certs.utils import helpers
-from sec_certs.utils.extract import normalize_match_string
-
-HEADERS = {
- "anssi": sec_certs.utils.extract.search_only_headers_anssi,
- "bsi": sec_certs.utils.extract.search_only_headers_bsi,
- "nscib": sec_certs.utils.extract.search_only_headers_nscib,
- "niap": sec_certs.utils.extract.search_only_headers_niap,
- "canada": sec_certs.utils.extract.search_only_headers_canada,
-}
-
-
-class ReferenceType(Enum):
- DIRECT = "direct"
- INDIRECT = "indirect"
-
-
-class CommonCriteriaCert(
- Certificate["CommonCriteriaCert", "CommonCriteriaCert.Heuristics", "CommonCriteriaCert.PdfData"],
- PandasSerializableType,
- ComplexSerializableType,
-):
- """
- Data structure for common criteria certificate. Contains several inner classes that layer the data logic.
- Can be serialized into/from json (`ComplexSerializableType`) or pandas (`PandasSerializableType)`.
- Is basic element of `CCDataset`. The functionality is mostly related to holding data and transformations that
- the certificate can handle itself. `CCDataset` class then instrument this functionality.
- """
-
- cc_url = "http://www.commoncriteriaportal.org"
- empty_st_url = "http://www.commoncriteriaportal.org/files/epfiles/"
-
- @dataclass(eq=True, frozen=True)
- class MaintenanceReport(ComplexSerializableType):
- """
- Object for holding maintenance reports.
- """
-
- maintenance_date: date | None
- maintenance_title: str | None
- maintenance_report_link: str | None
- maintenance_st_link: str | None
-
- def __post_init__(self):
- super().__setattr__(
- "maintenance_report_link", sec_certs.utils.sanitization.sanitize_link(self.maintenance_report_link)
- )
- super().__setattr__(
- "maintenance_st_link", sec_certs.utils.sanitization.sanitize_link(self.maintenance_st_link)
- )
- super().__setattr__(
- "maintenance_title", sec_certs.utils.sanitization.sanitize_string(self.maintenance_title)
- )
- super().__setattr__("maintenance_date", sec_certs.utils.sanitization.sanitize_date(self.maintenance_date))
-
- @classmethod
- def from_dict(cls, dct: dict) -> CommonCriteriaCert.MaintenanceReport:
- new_dct = dct.copy()
- new_dct["maintenance_date"] = (
- date.fromisoformat(dct["maintenance_date"])
- if isinstance(dct["maintenance_date"], str)
- else dct["maintenance_date"]
- )
- return super().from_dict(new_dct)
-
- def __lt__(self, other):
- return self.maintenance_date < other.maintenance_date
-
- @dataclass(init=False)
- class InternalState(ComplexSerializableType):
- """
- Holds internal state of the certificate, whether downloads and converts of individual components succeeded. Also
- holds information about errors and paths to the files.
- """
-
- st_download_ok: bool # Whether target download went OK
- report_download_ok: bool # Whether report download went OK
- st_convert_garbage: bool # Whether initial target conversion resulted in garbage
- report_convert_garbage: bool # Whether initial report conversion resulted in garbage
- st_convert_ok: bool # Whether overall target conversion went OK (either pdftotext or via OCR)
- report_convert_ok: bool # Whether overall report conversion went OK (either pdftotext or via OCR)
- st_extract_ok: bool # Whether target extraction went OK
- report_extract_ok: bool # Whether report extraction went OK
-
- st_pdf_hash: str | None
- report_pdf_hash: str | None
- st_txt_hash: str | None
- report_txt_hash: str | None
-
- st_pdf_path: Path
- report_pdf_path: Path
- st_txt_path: Path
- report_txt_path: Path
-
- def __init__(
- self,
- st_download_ok: bool = False,
- report_download_ok: bool = False,
- st_convert_garbage: bool = False,
- report_convert_garbage: bool = False,
- st_convert_ok: bool = False,
- report_convert_ok: bool = False,
- st_extract_ok: bool = False,
- report_extract_ok: bool = False,
- st_pdf_hash: str | None = None,
- report_pdf_hash: str | None = None,
- st_txt_hash: str | None = None,
- report_txt_hash: str | None = None,
- ):
- super().__init__()
- self.st_download_ok = st_download_ok
- self.report_download_ok = report_download_ok
- self.st_convert_garbage = st_convert_garbage
- self.report_convert_garbage = report_convert_garbage
- self.st_convert_ok = st_convert_ok
- self.report_convert_ok = report_convert_ok
- self.st_extract_ok = st_extract_ok
- self.report_extract_ok = report_extract_ok
- self.st_pdf_hash = st_pdf_hash
- self.report_pdf_hash = report_pdf_hash
- self.st_txt_hash = st_txt_hash
- self.report_txt_hash = report_txt_hash
-
- @property
- def serialized_attributes(self) -> list[str]:
- return [
- "st_download_ok",
- "report_download_ok",
- "st_convert_garbage",
- "report_convert_garbage",
- "st_convert_ok",
- "report_convert_ok",
- "st_extract_ok",
- "report_extract_ok",
- "st_pdf_hash",
- "report_pdf_hash",
- "st_txt_hash",
- "report_txt_hash",
- ]
-
- def report_is_ok_to_download(self, fresh: bool = True) -> bool:
- return True if fresh else not self.report_download_ok
-
- def st_is_ok_to_download(self, fresh: bool = True) -> bool:
- return True if fresh else not self.st_download_ok
-
- def report_is_ok_to_convert(self, fresh: bool = True) -> bool:
- return self.report_download_ok if fresh else self.report_download_ok and not self.report_convert_ok
-
- def st_is_ok_to_convert(self, fresh: bool = True) -> bool:
- return self.st_download_ok if fresh else self.st_download_ok and not self.st_convert_ok
-
- def report_is_ok_to_analyze(self, fresh: bool = True) -> bool:
- if fresh is True:
- return self.report_download_ok and self.report_convert_ok
- else:
- return self.report_download_ok and self.report_convert_ok and not self.report_extract_ok
-
- def st_is_ok_to_analyze(self, fresh: bool = True) -> bool:
- if fresh is True:
- return self.st_download_ok and self.st_convert_ok
- else:
- return self.st_download_ok and self.st_convert_ok and not self.st_extract_ok
-
- @dataclass
- class PdfData(BasePdfData, ComplexSerializableType):
- """
- Class that holds data extracted from pdf files.
- """
-
- report_metadata: dict[str, Any] | None = field(default=None)
- st_metadata: dict[str, Any] | None = field(default=None)
- report_frontpage: dict[str, dict[str, Any]] | None = field(default=None)
- st_frontpage: dict[str, dict[str, Any]] | None = field(default=None)
- report_keywords: dict[str, Any] | None = field(default=None)
- st_keywords: dict[str, Any] | None = field(default=None)
- report_filename: str | None = field(default=None)
- st_filename: str | None = field(default=None)
-
- def __bool__(self) -> bool:
- return any([x is not None for x in vars(self)])
-
- @property
- def bsi_data(self) -> dict[str, Any] | None:
- """
- Returns frontpage data related to BSI-provided information
- """
- return self.report_frontpage.get("bsi", None) if self.report_frontpage else None
-
- @property
- def niap_data(self) -> dict[str, Any] | None:
- """
- Returns frontpage data related to niap-provided information
- """
- return self.report_frontpage.get("niap", None) if self.report_frontpage else None
-
- @property
- def nscib_data(self) -> dict[str, Any] | None:
- """
- Returns frontpage data related to nscib-provided information
- """
- return self.report_frontpage.get("nscib", None) if self.report_frontpage else None
-
- @property
- def canada_data(self) -> dict[str, Any] | None:
- """
- Returns frontpage data related to canada-provided information
- """
- return self.report_frontpage.get("canada", None) if self.report_frontpage else None
-
- @property
- def anssi_data(self) -> dict[str, Any] | None:
- """
- Returns frontpage data related to ANSSI-provided information
- """
- return self.report_frontpage.get("anssi", None) if self.report_frontpage else None
-
- @property
- def cert_lab(self) -> list[str] | None:
- """
- Returns labs for which certificate data was parsed.
- """
- labs = [
- data["cert_lab"].split(" ")[0].upper()
- for data in [self.bsi_data, self.anssi_data, self.niap_data, self.nscib_data, self.canada_data]
- if data
- ]
- return labs if labs else None
-
- @property
- def bsi_cert_id(self) -> str | None:
- return self.bsi_data.get("cert_id", None) if self.bsi_data else None
-
- @property
- def niap_cert_id(self) -> str | None:
- return self.niap_data.get("cert_id", None) if self.niap_data else None
-
- @property
- def nscib_cert_id(self) -> str | None:
- return self.nscib_data.get("cert_id", None) if self.nscib_data else None
-
- @property
- def canada_cert_id(self) -> str | None:
- return self.canada_data.get("cert_id", None) if self.canada_data else None
-
- @property
- def anssi_cert_id(self) -> str | None:
- return self.anssi_data.get("cert_id", None) if self.anssi_data else None
-
- def frontpage_cert_id(self, scheme: str) -> dict[str, float]:
- """
- Get cert_id candidate from the frontpage of the report.
- """
- scheme_map = {
- "DE": self.bsi_cert_id,
- "US": self.niap_cert_id,
- "NL": self.nscib_cert_id,
- "CA": self.canada_cert_id,
- "FR": self.anssi_cert_id,
- }
- if scheme in scheme_map and (candidate := scheme_map[scheme]):
- return {candidate: 1.0}
- return {}
-
- def filename_cert_id(self, scheme: str) -> dict[str, float]:
- """
- Get cert_id candidates from the matches in the report filename.
- """
- if not self.report_filename:
- return {}
- scheme_rules = rules["cc_cert_id"][scheme]
- matches: Counter = Counter()
- for rule in scheme_rules:
- match = re.search(rule, self.report_filename)
- if match:
- cert_id = normalize_match_string(match.group())
- matches[cert_id] += 1
- if not matches:
- return {}
- total = max(matches.values())
- results = {}
- for candidate, count in matches.items():
- results[candidate] = count / total
- # TODO count length in weight
- return results
-
- def keywords_cert_id(self, scheme: str) -> dict[str, float]:
- """
- Get cert_id candidates from the keywords matches in the report.
- """
- if not self.report_keywords:
- return {}
- cert_id_matches = self.report_keywords.get("cc_cert_id")
- if not cert_id_matches:
- return {}
-
- if scheme not in cert_id_matches:
- return {}
- matches: Counter = Counter(cert_id_matches[scheme])
- if not matches:
- return {}
- total = max(matches.values())
- results = {}
- for candidate, count in matches.items():
- results[candidate] = count / total
- # TODO count length in weight
- return results
-
- def metadata_cert_id(self, scheme: str) -> dict[str, float]:
- """
- Get cert_id candidates from the report metadata.
- """
- scheme_rules = rules["cc_cert_id"][scheme]
- fields = ("/Title", "/Subject")
- matches: Counter = Counter()
- for meta_field in fields:
- field_val = self.report_metadata.get(meta_field) if self.report_metadata else None
- if not field_val:
- continue
- for rule in scheme_rules:
- match = re.search(rule, field_val)
- if match:
- cert_id = normalize_match_string(match.group())
- matches[cert_id] += 1
- if not matches:
- return {}
- total = max(matches.values())
- results = {}
- for candidate, count in matches.items():
- results[candidate] = count / total
- # TODO count length in weight
- return results
-
- def candidate_cert_ids(self, scheme: str) -> dict[str, float]:
- frontpage_id = self.frontpage_cert_id(scheme)
- metadata_id = self.metadata_cert_id(scheme)
- filename_id = self.filename_cert_id(scheme)
- keywords_id = self.keywords_cert_id(scheme)
-
- # Join them and weigh them, each is normalized with weights from 0 to 1 (if anything is returned)
- candidates: dict[str, float] = defaultdict(lambda: 0.0)
- # TODO: Add heuristic based on ordering of ids (and extracted year + increment)
- # TODO: Add heuristic based on length
- for candidate, count in frontpage_id.items():
- candidates[canonicalize(candidate, scheme)] += count * 1.5
- for candidate, count in metadata_id.items():
- candidates[canonicalize(candidate, scheme)] += count * 1.2
- for candidate, count in keywords_id.items():
- candidates[canonicalize(candidate, scheme)] += count * 1.0
- for candidate, count in filename_id.items():
- candidates[canonicalize(candidate, scheme)] += count * 1.0
- return candidates
-
- @dataclass
- class Heuristics(BaseHeuristics, ComplexSerializableType):
- """
- Class for various heuristics related to CommonCriteriaCert
- """
-
- extracted_versions: set[str] | None = field(default=None)
- cpe_matches: set[str] | None = field(default=None)
- verified_cpe_matches: set[str] | None = field(default=None)
- related_cves: set[str] | None = field(default=None)
- cert_lab: list[str] | None = field(default=None)
- cert_id: str | None = field(default=None)
- st_references: References = field(default_factory=References)
- report_references: References = field(default_factory=References)
- extracted_sars: set[SAR] | None = field(default=None)
- direct_transitive_cves: set[str] | None = field(default=None)
- indirect_transitive_cves: set[str] | None = field(default=None)
-
- @property
- def serialized_attributes(self) -> list[str]:
- return copy.deepcopy(super().serialized_attributes)
-
- pandas_columns: ClassVar[list[str]] = [
- "dgst",
- "cert_id",
- "name",
- "status",
- "category",
- "manufacturer",
- "scheme",
- "security_level",
- "eal",
- "not_valid_before",
- "not_valid_after",
- "report_link",
- "st_link",
- "cert_link",
- "manufacturer_web",
- "extracted_versions",
- "cpe_matches",
- "verified_cpe_matches",
- "related_cves",
- "directly_referenced_by",
- "indirectly_referenced_by",
- "directly_referencing",
- "indirectly_referencing",
- "extracted_sars",
- "protection_profiles",
- "cert_lab",
- ]
-
- def __init__(
- self,
- status: str,
- category: str,
- name: str,
- manufacturer: str | None,
- scheme: str,
- security_level: str | set[str],
- not_valid_before: date | None,
- not_valid_after: date | None,
- report_link: str,
- st_link: str,
- cert_link: str | None,
- manufacturer_web: str | None,
- protection_profiles: set[ProtectionProfile] | None,
- maintenance_updates: set[MaintenanceReport] | None,
- state: InternalState | None,
- pdf_data: PdfData | None,
- heuristics: Heuristics | None,
- ):
- super().__init__()
-
- self.status = status
- self.category = category
- self.name = sec_certs.utils.sanitization.sanitize_string(name)
-
- self.manufacturer = None
- if manufacturer:
- self.manufacturer = sec_certs.utils.sanitization.sanitize_string(manufacturer)
-
- self.scheme = scheme
- self.security_level = sec_certs.utils.sanitization.sanitize_security_levels(security_level)
- self.not_valid_before = sec_certs.utils.sanitization.sanitize_date(not_valid_before)
- self.not_valid_after = sec_certs.utils.sanitization.sanitize_date(not_valid_after)
- self.report_link = sec_certs.utils.sanitization.sanitize_link(report_link)
- self.st_link = sec_certs.utils.sanitization.sanitize_link(st_link)
- self.cert_link = sec_certs.utils.sanitization.sanitize_link(cert_link)
- self.manufacturer_web = sec_certs.utils.sanitization.sanitize_link(manufacturer_web)
- self.protection_profiles = protection_profiles
- self.maintenance_updates = maintenance_updates
- self.state = self.InternalState() if not state else state
- self.pdf_data = self.PdfData() if not pdf_data else pdf_data
- self.heuristics: CommonCriteriaCert.Heuristics = self.Heuristics() if not heuristics else heuristics
-
- @property
- def dgst(self) -> str:
- """
- Computes the primary key of the sample using first 16 bytes of SHA-256 digest
- """
- if not (self.name is not None and self.report_link is not None and self.category is not None):
- raise RuntimeError("Certificate digest can't be computed, because information is missing.")
- return helpers.get_first_16_bytes_sha256(self.category + self.name + self.report_link)
-
- @property
- def eal(self) -> str | None:
- """
- Returns EAL of certificate if it was extracted, None otherwise.
- """
- res = [x for x in self.security_level if re.match(security_level_csv_scan, x)]
- if res and len(res) == 1:
- return res[0]
- if res and len(res) > 1:
- raise ValueError(f"Expected single EAL in security_level field, got: {res}")
- else:
- if self.protection_profiles:
- return helpers.choose_lowest_eal({x.pp_eal for x in self.protection_profiles if x.pp_eal})
- else:
- return None
-
- @property
- def actual_sars(self) -> set[SAR] | None:
- """
- Computes actual SARs. First, SARs implied by EAL are computed. Then, these are augmented with heuristically extracted SARs
- :return Optional[Set[SAR]]: Set of actual SARs of a certificate, None if empty
- """
- sars = dict()
- if self.eal:
- sars = {x[0]: SAR(x[0], x[1]) for x in SARS_IMPLIED_FROM_EAL[self.eal[:4]]}
-
- if self.heuristics.extracted_sars:
- for sar in self.heuristics.extracted_sars:
- if sar not in sars or sar.level > sars[sar.family].level:
- sars[sar.family] = sar
-
- return set(sars.values()) if sars else None
-
- @property
- def label_studio_title(self) -> str | None:
- return self.name
-
- @property
- def pandas_tuple(self) -> tuple:
- """
- Returns tuple of attributes meant for pandas serialization
- """
- return (
- self.dgst,
- self.heuristics.cert_id,
- self.name,
- self.status,
- self.category,
- self.manufacturer,
- self.scheme,
- self.security_level,
- self.eal,
- self.not_valid_before,
- self.not_valid_after,
- self.report_link,
- self.st_link,
- self.cert_link,
- self.manufacturer_web,
- self.heuristics.extracted_versions,
- self.heuristics.cpe_matches,
- self.heuristics.verified_cpe_matches,
- self.heuristics.related_cves,
- self.heuristics.report_references.directly_referenced_by,
- self.heuristics.report_references.indirectly_referenced_by,
- self.heuristics.report_references.directly_referencing,
- self.heuristics.report_references.indirectly_referencing,
- self.heuristics.extracted_sars,
- [x.pp_name for x in self.protection_profiles] if self.protection_profiles else np.nan,
- self.heuristics.cert_lab[0] if (self.heuristics.cert_lab and self.heuristics.cert_lab[0]) else np.nan,
- )
-
- def __str__(self) -> str:
- printed_manufacturer = self.manufacturer if self.manufacturer else "Unknown manufacturer"
- return str(printed_manufacturer) + " " + str(self.name) + " dgst: " + self.dgst
-
- def merge(self, other: CommonCriteriaCert, other_source: str | None = None) -> None:
- """
- Merges with other CC sample. Assuming they come from different sources, e.g., csv and html.
- Assuming that html source has better protection profiles, they overwrite CSV info
- On other values the sanity checks are made.
- """
- if self != other:
- logger.warning(
- f"Attempting to merge divergent certificates: self[dgst]={self.dgst}, other[dgst]={other.dgst}"
- )
-
- for att, val in vars(self).items():
- if not val:
- setattr(self, att, getattr(other, att))
- elif other_source == "html" and att == "protection_profiles":
- setattr(self, att, getattr(other, att))
- elif other_source == "html" and att == "maintenance_updates":
- setattr(self, att, getattr(other, att))
- elif att == "state":
- setattr(self, att, getattr(other, att))
- else:
- if getattr(self, att) != getattr(other, att):
- logger.warning(
- f"When merging certificates with dgst {self.dgst}, the following mismatch occured: Attribute={att}, self[{att}]={getattr(self, att)}, other[{att}]={getattr(other, att)}"
- )
-
- @classmethod
- def from_dict(cls, dct: dict) -> CommonCriteriaCert:
- """
- Deserializes dictionary into `CommonCriteriaCert`
- """
- new_dct = dct.copy()
- new_dct["maintenance_updates"] = set(dct["maintenance_updates"])
- new_dct["protection_profiles"] = set(dct["protection_profiles"])
- new_dct["not_valid_before"] = (
- date.fromisoformat(dct["not_valid_before"])
- if isinstance(dct["not_valid_before"], str)
- else dct["not_valid_before"]
- )
- new_dct["not_valid_after"] = (
- date.fromisoformat(dct["not_valid_after"])
- if isinstance(dct["not_valid_after"], str)
- else dct["not_valid_after"]
- )
- return super(cls, CommonCriteriaCert).from_dict(new_dct)
-
- @staticmethod
- def _html_row_get_name(cell: Tag) -> str:
- return list(cell.stripped_strings)[0]
-
- @staticmethod
- def _html_row_get_manufacturer(cell: Tag) -> str | None:
- if lst := list(cell.stripped_strings):
- return lst[0]
- else:
- return None
-
- @staticmethod
- def _html_row_get_scheme(cell: Tag) -> str:
- return list(cell.stripped_strings)[0]
-
- @staticmethod
- def _html_row_get_security_level(cell: Tag) -> set:
- return set(cell.stripped_strings)
-
- @staticmethod
- def _html_row_get_manufacturer_web(cell: Tag) -> str | None:
- for link in cell.find_all("a"):
- if link is not None and link.get("title") == "Vendor's web site" and link.get("href") != "http://":
- return link.get("href")
- return None
-
- @staticmethod
- def _html_row_get_protection_profiles(cell: Tag) -> set:
- protection_profiles = set()
- for link in list(cell.find_all("a")):
- if link.get("href") is not None and "/ppfiles/" in link.get("href"):
- protection_profiles.add(
- ProtectionProfile(
- pp_name=str(link.contents[0]), pp_eal=None, pp_link=CommonCriteriaCert.cc_url + link.get("href")
- )
- )
- return protection_profiles
-
- @staticmethod
- def _html_row_get_date(cell: Tag) -> date | None:
- text = cell.get_text()
- extracted_date = datetime.strptime(text, "%Y-%m-%d").date() if text else None
- return extracted_date
-
- @staticmethod
- def _html_row_get_report_st_links(cell: Tag) -> tuple[str, str]:
- links = cell.find_all("a")
- assert links[1].get("title").startswith("Certification Report")
- assert links[2].get("title").startswith("Security Target")
-
- report_link = CommonCriteriaCert.cc_url + links[1].get("href")
- security_target_link = CommonCriteriaCert.cc_url + links[2].get("href")
-
- return report_link, security_target_link
-
- @staticmethod
- def _html_row_get_cert_link(cell: Tag) -> str | None:
- links = cell.find_all("a")
- return CommonCriteriaCert.cc_url + links[0].get("href") if links else None
-
- @staticmethod
- def _html_row_get_maintenance_div(cell: Tag) -> Tag | None:
- divs = cell.find_all("div")
- for d in divs:
- if d.find("div") and d.stripped_strings and list(d.stripped_strings)[0] == "Maintenance Report(s)":
- return d
- return None
-
- @staticmethod
- def _html_row_get_maintenance_updates(main_div: Tag) -> set[CommonCriteriaCert.MaintenanceReport]:
- possible_updates = list(main_div.find_all("li"))
- maintenance_updates = set()
- for u in possible_updates:
- text = list(u.stripped_strings)[0]
- main_date = datetime.strptime(text.split(" ")[0], "%Y-%m-%d").date() if text else None
- main_title = text.split("– ")[1]
- main_report_link = None
- main_st_link = None
- links = u.find_all("a")
- for link in links:
- if link.get("title").startswith("Maintenance Report:"):
- main_report_link = CommonCriteriaCert.cc_url + link.get("href")
- elif link.get("title").startswith("Maintenance ST"):
- main_st_link = CommonCriteriaCert.cc_url + link.get("href")
- else:
- logger.error("Unknown link in Maintenance part!")
- maintenance_updates.add(
- CommonCriteriaCert.MaintenanceReport(main_date, main_title, main_report_link, main_st_link)
- )
- return maintenance_updates
-
- @classmethod
- def from_html_row(cls, row: Tag, status: str, category: str) -> CommonCriteriaCert:
- """
- Creates a CC sample from html row of commoncriteria.org webpage.
- """
-
- cells = list(row.find_all("td"))
- if len(cells) != 7:
- raise ValueError(f"Unexpected number of <td> elements in CC html row. Expected: 7, actual: {len(cells)}")
-
- name = CommonCriteriaCert._html_row_get_name(cells[0])
- manufacturer = CommonCriteriaCert._html_row_get_manufacturer(cells[1])
- manufacturer_web = CommonCriteriaCert._html_row_get_manufacturer_web(cells[1])
- scheme = CommonCriteriaCert._html_row_get_scheme(cells[6])
- security_level = CommonCriteriaCert._html_row_get_security_level(cells[5])
- protection_profiles = CommonCriteriaCert._html_row_get_protection_profiles(cells[0])
- not_valid_before = CommonCriteriaCert._html_row_get_date(cells[3])
- not_valid_after = CommonCriteriaCert._html_row_get_date(cells[4])
- report_link, st_link = CommonCriteriaCert._html_row_get_report_st_links(cells[0])
- cert_link = CommonCriteriaCert._html_row_get_cert_link(cells[2])
- maintenance_div = CommonCriteriaCert._html_row_get_maintenance_div(cells[0])
- maintenances = (
- CommonCriteriaCert._html_row_get_maintenance_updates(maintenance_div) if maintenance_div else set()
- )
-
- return cls(
- status,
- category,
- name,
- manufacturer,
- scheme,
- security_level,
- not_valid_before,
- not_valid_after,
- report_link,
- st_link,
- cert_link,
- manufacturer_web,
- protection_profiles,
- maintenances,
- None,
- None,
- None,
- )
-
- def set_local_paths(
- self,
- report_pdf_dir: str | Path | None,
- st_pdf_dir: str | Path | None,
- report_txt_dir: str | Path | None,
- st_txt_dir: str | Path | None,
- ) -> None:
- """
- Sets paths to files given the requested directories
-
- :param Optional[Union[str, Path]] report_pdf_dir: Directory where pdf reports shall be stored
- :param Optional[Union[str, Path]] st_pdf_dir: Directory where pdf security targets shall be stored
- :param Optional[Union[str, Path]] report_txt_dir: Directory where txt reports shall be stored
- :param Optional[Union[str, Path]] st_txt_dir: Directory where txt security targets shall be stored
- """
- if report_pdf_dir is not None:
- self.state.report_pdf_path = Path(report_pdf_dir) / (self.dgst + ".pdf")
- if st_pdf_dir is not None:
- self.state.st_pdf_path = Path(st_pdf_dir) / (self.dgst + ".pdf")
- if report_txt_dir is not None:
- self.state.report_txt_path = Path(report_txt_dir) / (self.dgst + ".txt")
- if st_txt_dir is not None:
- self.state.st_txt_path = Path(st_txt_dir) / (self.dgst + ".txt")
-
- @staticmethod
- def download_pdf_report(cert: CommonCriteriaCert) -> CommonCriteriaCert:
- """
- Downloads pdf of certification report given the certificate. Staticmethod to allow for parallelization.
-
- :param CommonCriteriaCert cert: cert to download the pdf report for
- :return CommonCriteriaCert: returns the modified certificate with updated state
- """
- exit_code: str | int
- if not cert.report_link:
- exit_code = "No link"
- else:
- exit_code = helpers.download_file(cert.report_link, cert.state.report_pdf_path)
- if exit_code != requests.codes.ok:
- error_msg = f"failed to download report from {cert.report_link}, code: {exit_code}"
- logger.error(f"Cert dgst: {cert.dgst} " + error_msg)
- cert.state.report_download_ok = False
- else:
- cert.state.report_download_ok = True
- cert.state.report_pdf_hash = helpers.get_sha256_filepath(cert.state.report_pdf_path)
- cert.pdf_data.report_filename = unquote_plus(str(urlparse(cert.report_link).path).split("/")[-1])
- return cert
-
- @staticmethod
- def download_pdf_st(cert: CommonCriteriaCert) -> CommonCriteriaCert:
- """
- Downloads pdf of security target given the certificate. Staticmethod to allow for parallelization.
-
- :param CommonCriteriaCert cert: cert to download the pdf security target for
- :return CommonCriteriaCert: returns the modified certificate with updated state
- """
- exit_code: str | int
- if not cert.st_link:
- exit_code = "No link"
- else:
- exit_code = helpers.download_file(cert.st_link, cert.state.st_pdf_path)
- if exit_code != requests.codes.ok:
- error_msg = f"failed to download ST from {cert.st_link}, code: {exit_code}"
- logger.error(f"Cert dgst: {cert.dgst} " + error_msg)
- cert.state.st_download_ok = False
- else:
- cert.state.st_download_ok = True
- cert.state.st_pdf_hash = helpers.get_sha256_filepath(cert.state.st_pdf_path)
- cert.pdf_data.st_filename = unquote_plus(str(urlparse(cert.st_link).path).split("/")[-1])
- return cert
-
- @staticmethod
- def convert_report_pdf(cert: CommonCriteriaCert) -> CommonCriteriaCert:
- """
- Converts the pdf certification report to txt, given the certificate. Staticmethod to allow for parallelization.
-
- :param CommonCriteriaCert cert: cert to download the pdf report for
- :return CommonCriteriaCert: the modified certificate with updated state
- """
- ocr_done, ok_result = sec_certs.utils.pdf.convert_pdf_file(
- cert.state.report_pdf_path, cert.state.report_txt_path
- )
- # If OCR was done the result was garbage
- cert.state.report_convert_garbage = ocr_done
- # And put the whole result into convert_ok
- cert.state.report_convert_ok = ok_result
- if not ok_result:
- error_msg = "failed to convert report pdf->txt"
- logger.error(f"Cert dgst: {cert.dgst} " + error_msg)
- else:
- cert.state.report_txt_hash = helpers.get_sha256_filepath(cert.state.report_txt_path)
- return cert
-
- @staticmethod
- def convert_st_pdf(cert: CommonCriteriaCert) -> CommonCriteriaCert:
- """
- Converts the pdf security target to txt, given the certificate. Staticmethod to allow for parallelization.
-
- :param CommonCriteriaCert cert: cert to download the pdf security target for
- :return CommonCriteriaCert: the modified certificate with updated state
- """
- ocr_done, ok_result = sec_certs.utils.pdf.convert_pdf_file(cert.state.st_pdf_path, cert.state.st_txt_path)
- # If OCR was done the result was garbage
- cert.state.st_convert_garbage = ocr_done
- # And put the whole result into convert_ok
- cert.state.st_convert_ok = ok_result
- if not ok_result:
- error_msg = "failed to convert security target pdf->txt"
- logger.error(f"Cert dgst: {cert.dgst} " + error_msg)
- else:
- cert.state.st_txt_hash = helpers.get_sha256_filepath(cert.state.st_txt_path)
- return cert
-
- @staticmethod
- def extract_st_pdf_metadata(cert: CommonCriteriaCert) -> CommonCriteriaCert:
- """
- Extracts metadata from security target pdf given the certificate. Staticmethod to allow for parallelization.
-
- :param CommonCriteriaCert cert: cert to extract the metadata for.
- :return CommonCriteriaCert: the modified certificate with updated state
- """
- response, cert.pdf_data.st_metadata = sec_certs.utils.pdf.extract_pdf_metadata(cert.state.st_pdf_path)
- if response != constants.RETURNCODE_OK:
- cert.state.st_extract_ok = False
- else:
- cert.state.st_extract_ok = True
- return cert
-
- @staticmethod
- def extract_report_pdf_metadata(cert: CommonCriteriaCert) -> CommonCriteriaCert:
- """
- Extracts metadata from certification report pdf given the certificate. Staticmethod to allow for parallelization.
-
- :param CommonCriteriaCert cert: cert to extract the metadata for.
- :return CommonCriteriaCert: the modified certificate with updated state
- """
- response, cert.pdf_data.report_metadata = sec_certs.utils.pdf.extract_pdf_metadata(cert.state.report_pdf_path)
- if response != constants.RETURNCODE_OK:
- cert.state.report_extract_ok = False
- else:
- cert.state.report_extract_ok = True
- return cert
-
- @staticmethod
- def extract_st_pdf_frontpage(cert: CommonCriteriaCert) -> CommonCriteriaCert:
- """
- Extracts data from security target pdf frontpage given the certificate. Staticmethod to allow for parallelization.
-
- :param CommonCriteriaCert cert: cert to extract the frontpage data for.
- :return CommonCriteriaCert: the modified certificate with updated state
- """
- cert.pdf_data.st_frontpage = {}
-
- for header_type, associated_header_func in HEADERS.items():
- response, cert.pdf_data.st_frontpage[header_type] = associated_header_func(cert.state.st_txt_path)
-
- if response != constants.RETURNCODE_OK:
- cert.state.st_extract_ok = False
- return cert
-
- @staticmethod
- def extract_report_pdf_frontpage(cert: CommonCriteriaCert) -> CommonCriteriaCert:
- """
- Extracts data from certification report pdf frontpage given the certificate. Staticmethod to allow for parallelization.
-
- :param CommonCriteriaCert cert: cert to extract the frontpage data for.
- :return CommonCriteriaCert: the modified certificate with updated state
- """
- cert.pdf_data.report_frontpage = {}
-
- for header_type, associated_header_func in HEADERS.items():
- response, cert.pdf_data.report_frontpage[header_type] = associated_header_func(cert.state.report_txt_path)
-
- if response != constants.RETURNCODE_OK:
- cert.state.report_extract_ok = False
- return cert
-
- @staticmethod
- def extract_report_pdf_keywords(cert: CommonCriteriaCert) -> CommonCriteriaCert:
- """
- Matches regular expresions in txt obtained from certification report and extracts the matches into attribute.
- Static method to allow for parallelization
-
- :param CommonCriteriaCert cert: certificate to extract the keywords for.
- :return CommonCriteriaCert: the modified certificate with extracted keywords.
- """
- report_keywords = sec_certs.utils.extract.extract_keywords(cert.state.report_txt_path, cc_rules)
- if report_keywords is None:
- cert.state.report_extract_ok = False
- else:
- cert.pdf_data.report_keywords = report_keywords
- return cert
-
- @staticmethod
- def extract_st_pdf_keywords(cert: CommonCriteriaCert) -> CommonCriteriaCert:
- """
- Matches regular expresions in txt obtained from security target and extracts the matches into attribute.
- Static method to allow for parallelization
-
- :param CommonCriteriaCert cert: certificate to extract the keywords for.
- :return CommonCriteriaCert: the modified certificate with extracted keywords.
- """
- st_keywords = sec_certs.utils.extract.extract_keywords(cert.state.st_txt_path, cc_rules)
- if st_keywords is None:
- cert.state.st_extract_ok = False
- else:
- cert.pdf_data.st_keywords = st_keywords
- return cert
-
- def compute_heuristics_version(self) -> None:
- """
- Fills in the heuristically obtained version of certified product into attribute in heuristics class.
- """
- self.heuristics.extracted_versions = helpers.compute_heuristics_version(self.name) if self.name else set()
-
- def compute_heuristics_cert_lab(self) -> None:
- """
- Fills in the heuristically obtained evaluation laboratory into attribute in heuristics class.
- """
- if not self.pdf_data:
- logger.error("Cannot compute sample lab when pdf files were not processed.")
- return
- self.heuristics.cert_lab = self.pdf_data.cert_lab
-
- def compute_heuristics_cert_id(self):
- """
- Compute the heuristics cert_id of this cert, using several methods.
-
- The candidate cert_ids are extracted from the frontpage, PDF metadata, filename, and keywords matches.
-
- Finally, the cert_id is canonicalized.
- """
- if not self.pdf_data:
- logger.warning("Cannot compute sample id when pdf files were not processed.")
- return
- # Extract candidate cert_ids
- candidates = self.pdf_data.candidate_cert_ids(self.scheme)
-
- if candidates:
- max_weight = max(candidates.values())
- max_candidates = list(filter(lambda x: candidates[x] == max_weight, candidates.keys()))
- max_candidates.sort(key=len, reverse=True)
- self.heuristics.cert_id = max_candidates[0]
diff --git a/sec_certs/sample/cpe.py b/sec_certs/sample/cpe.py
deleted file mode 100644
index a7532f7f..00000000
--- a/sec_certs/sample/cpe.py
+++ /dev/null
@@ -1,100 +0,0 @@
-from __future__ import annotations
-
-from dataclasses import dataclass
-from functools import lru_cache
-from typing import Any, ClassVar
-
-from sec_certs import constants
-from sec_certs.serialization.json import ComplexSerializableType
-from sec_certs.serialization.pandas import PandasSerializableType
-from sec_certs.utils import helpers
-
-
-@dataclass(init=False)
-class CPE(PandasSerializableType, ComplexSerializableType):
- uri: str
- version: str
- vendor: str
- item_name: str
- title: str | None
- start_version: tuple[str, str] | None
- end_version: tuple[str, str] | None
-
- __slots__ = ["uri", "version", "vendor", "item_name", "title", "start_version", "end_version"]
-
- pandas_columns: ClassVar[list[str]] = [
- "uri",
- "vendor",
- "item_name",
- "version",
- "title",
- ]
-
- def __init__(
- self,
- uri: str,
- title: str | None = None,
- start_version: tuple[str, str] | None = None,
- end_version: tuple[str, str] | None = None,
- ):
- super().__init__()
- self.uri = uri
-
- splitted = helpers.split_unescape(self.uri, ":")
- self.vendor = " ".join(splitted[3].split("_"))
- self.item_name = " ".join(splitted[4].split("_"))
- self.version = self.normalize_version(" ".join(splitted[5].split("_")))
- self.title = title
- self.start_version = start_version
- self.end_version = end_version
-
- def __lt__(self, other: CPE) -> bool:
- return self.uri < other.uri
-
- @staticmethod
- def normalize_version(version: str) -> str:
- """
- Maps common empty versions (empty '', asterisk '*') to unified empty version (constants.CPE_VERSION_NA)
- """
- if version in {"", "*"}:
- return constants.CPE_VERSION_NA
- return version
-
- @classmethod
- def from_dict(cls, dct: dict[str, Any]) -> CPE:
- if isinstance(dct["start_version"], list):
- dct["start_version"] = tuple(dct["start_version"])
- if isinstance(dct["end_version"], list):
- dct["end_version"] = tuple(dct["end_version"])
- return super().from_dict(dct)
-
- @property
- def serialized_attributes(self) -> list[str]:
- return ["uri", "title", "start_version", "end_version"]
-
- @property
- def update(self) -> str:
- if self.uri is None:
- raise RuntimeError("URI is missing.")
- return " ".join(self.uri.split(":")[6].split("_"))
-
- @property
- def target_hw(self) -> str:
- if self.uri is None:
- raise RuntimeError("URI is missing.")
- return " ".join(self.uri.split(":")[10].split("_"))
-
- @property
- def pandas_tuple(self) -> tuple:
- return self.uri, self.vendor, self.item_name, self.version, self.title
-
- def __hash__(self) -> int:
- return hash((self.uri, self.start_version, self.end_version))
-
- def __eq__(self, other: object) -> bool:
- return isinstance(other, self.__class__) and self.uri == other.uri
-
-
-@lru_cache(maxsize=4096)
-def cached_cpe(*args, **kwargs):
- return CPE(*args, **kwargs)
diff --git a/sec_certs/sample/cve.py b/sec_certs/sample/cve.py
deleted file mode 100644
index ec024d35..00000000
--- a/sec_certs/sample/cve.py
+++ /dev/null
@@ -1,185 +0,0 @@
-from __future__ import annotations
-
-import datetime
-import itertools
-from dataclasses import dataclass
-from typing import Any, ClassVar
-
-from dateutil.parser import isoparse
-
-from sec_certs.sample.cpe import CPE, cached_cpe
-from sec_certs.serialization.json import ComplexSerializableType
-from sec_certs.serialization.pandas import PandasSerializableType
-
-
-@dataclass(init=False)
-class CVE(PandasSerializableType, ComplexSerializableType):
- @dataclass(eq=True)
- class Impact(ComplexSerializableType):
- base_score: float
- severity: str
- exploitability_score: float
- impact_score: float
-
- __slots__ = ["base_score", "severity", "exploitability_score", "impact_score"]
-
- @classmethod
- def from_nist_dict(cls, dct: dict[str, Any]) -> CVE.Impact:
- """
- Will load Impact from dictionary defined at https://nvd.nist.gov/feeds/json/cve/1.1
- """
- if not dct["impact"]:
- return cls(0, "", 0, 0)
- elif "baseMetricV3" in dct["impact"]:
- return cls(
- dct["impact"]["baseMetricV3"]["cvssV3"]["baseScore"],
- dct["impact"]["baseMetricV3"]["cvssV3"]["baseSeverity"],
- dct["impact"]["baseMetricV3"]["exploitabilityScore"],
- dct["impact"]["baseMetricV3"]["impactScore"],
- )
- elif "baseMetricV2" in dct["impact"]:
- return cls(
- dct["impact"]["baseMetricV2"]["cvssV2"]["baseScore"],
- dct["impact"]["baseMetricV2"]["severity"],
- dct["impact"]["baseMetricV2"]["exploitabilityScore"],
- dct["impact"]["baseMetricV2"]["impactScore"],
- )
- raise ValueError("NIST Dict for CVE Impact badly formatted.")
-
- cve_id: str
- vulnerable_cpes: list[CPE]
- impact: Impact
- published_date: datetime.datetime | None
- cwe_ids: set[str] | None
-
- __slots__ = ["cve_id", "vulnerable_cpes", "impact", "published_date", "cwe_ids"]
-
- pandas_columns: ClassVar[list[str]] = [
- "cve_id",
- "vulnerable_cpes",
- "base_score",
- "severity",
- "explotability_score",
- "impact_score",
- "published_date",
- "cwe_ids",
- ]
-
- def __init__(
- self, cve_id: str, vulnerable_cpes: list[CPE], impact: Impact, published_date: str, cwe_ids: set[str] | None
- ):
- super().__init__()
- self.cve_id = cve_id
- self.vulnerable_cpes = vulnerable_cpes
- self.impact = impact
- self.published_date = isoparse(published_date)
- self.cwe_ids = cwe_ids
-
- def __hash__(self) -> int:
- return hash(self.cve_id)
-
- def __eq__(self, other: object) -> bool:
- if not isinstance(other, CVE):
- return False
- return self.cve_id == other.cve_id
-
- def __lt__(self, other: object) -> bool:
- if not isinstance(other, CVE):
- raise ValueError(f"Cannot compare CVE with {type(other)} type.")
- self_year = int(self.cve_id.split("-")[1])
- self_id = int(self.cve_id.split("-")[2])
- other_year = int(other.cve_id.split("-")[1])
- other_id = int(other.cve_id.split("-")[2])
-
- return self_year < other_year if self_year != other_year else self_id < other_id
-
- @property
- def pandas_tuple(self):
- return (
- self.cve_id,
- self.vulnerable_cpes,
- self.impact.base_score,
- self.impact.severity,
- self.impact.exploitability_score,
- self.impact.impact_score,
- self.published_date,
- self.cwe_ids,
- )
-
- def to_dict(self) -> dict[str, Any]:
- return {
- "cve_id": self.cve_id,
- "vulnerable_cpes": self.vulnerable_cpes,
- "impact": self.impact,
- "published_date": self.published_date.isoformat() if self.published_date else None,
- "cwe_ids": self.cwe_ids,
- }
-
- @staticmethod
- def _parse_nist_dict(lst: list) -> list[CPE]:
- cpes: list[CPE] = []
-
- for x in lst:
- if x["vulnerable"]:
- cpe_uri = x["cpe23Uri"]
- version_start: tuple[str, str] | None
- version_end: tuple[str, str] | None
- if "versionStartIncluding" in x and x["versionStartIncluding"]:
- version_start = ("including", x["versionStartIncluding"])
- elif "versionStartExcluding" in x and x["versionStartExcluding"]:
- version_start = ("excluding", x["versionStartExcluding"])
- else:
- version_start = None
-
- if "versionEndIncluding" in x and x["versionEndIncluding"]:
- version_end = ("including", x["versionEndIncluding"])
- elif "versionEndExcluding" in x and x["versionEndExcluding"]:
- version_end = ("excluding", x["versionEndExcluding"])
- else:
- version_end = None
-
- cpes.append(cached_cpe(cpe_uri, start_version=version_start, end_version=version_end))
-
- return cpes
-
- @classmethod
- def from_nist_dict(cls, dct: dict) -> CVE:
- """
- Will load CVE from dictionary defined at https://nvd.nist.gov/feeds/json/cve/1.1
- """
-
- def get_vulnerable_cpes_from_nist_dict(dct: dict) -> list[CPE]:
- def get_vulnerable_cpes_from_node(node: dict) -> list[CPE]:
- cpes: list[CPE] = []
-
- if node["operator"] == "AND":
- return cpes
-
- if "children" in node:
- for child in node["children"]:
- cpes += get_vulnerable_cpes_from_node(child)
-
- if "cpe_match" not in node:
- return cpes
-
- candidates = node["cpe_match"]
- cpes += CVE._parse_nist_dict(candidates)
-
- return cpes
-
- return list(
- itertools.chain.from_iterable(get_vulnerable_cpes_from_node(x) for x in dct["configurations"]["nodes"])
- )
-
- cve_id = dct["cve"]["CVE_data_meta"]["ID"]
- impact = cls.Impact.from_nist_dict(dct)
- vulnerable_cpes = get_vulnerable_cpes_from_nist_dict(dct)
- published_date = dct["publishedDate"]
- cwe_ids = cls.parse_cwe_data(dct)
-
- return cls(cve_id, vulnerable_cpes, impact, published_date, cwe_ids)
-
- @staticmethod
- def parse_cwe_data(dct: dict) -> set[str] | None:
- descriptions = dct["cve"]["problemtype"]["problemtype_data"][0]["description"]
- return {x["value"] for x in descriptions} if descriptions else None
diff --git a/sec_certs/sample/fips.py b/sec_certs/sample/fips.py
deleted file mode 100644
index 770b21b0..00000000
--- a/sec_certs/sample/fips.py
+++ /dev/null
@@ -1,649 +0,0 @@
-from __future__ import annotations
-
-import itertools
-import re
-from dataclasses import dataclass, field
-from datetime import date
-from pathlib import Path
-from typing import Any, Callable, ClassVar, Final, Literal
-
-import dateutil
-import numpy as np
-import requests
-from bs4 import BeautifulSoup, Tag
-from tabula import read_pdf
-
-import sec_certs.constants as constants
-import sec_certs.utils.extract
-import sec_certs.utils.helpers as helpers
-import sec_certs.utils.pdf
-import sec_certs.utils.pdf as pdf
-import sec_certs.utils.tables as tables
-from sec_certs.cert_rules import FIPS_ALGS_IN_TABLE, fips_rules
-from sec_certs.config.configuration import config
-from sec_certs.sample.certificate import Certificate
-from sec_certs.sample.certificate import Heuristics as BaseHeuristics
-from sec_certs.sample.certificate import PdfData as BasePdfData
-from sec_certs.sample.certificate import References, logger
-from sec_certs.sample.cpe import CPE
-from sec_certs.serialization.json import ComplexSerializableType
-from sec_certs.serialization.pandas import PandasSerializableType
-from sec_certs.utils.helpers import fips_dgst
-
-
-class FIPSHTMLParser:
- def __init__(self, soup: BeautifulSoup):
- self._soup = soup
-
- def get_web_data_and_algorithms(self) -> tuple[set[str], FIPSCertificate.WebData]:
- divs = self._soup.find_all("div", class_="panel panel-default")
- details_div, vendor_div, related_files_div, validation_history_div = divs
- details_dict = self._build_details_dict(details_div)
-
- vendor_dict = self._build_vendor_dict(vendor_div)
- related_files_dict = self._build_related_files_dict(related_files_div)
- validation_history_dict = self._build_validation_history_dict(validation_history_div)
-
- algorithms = set()
- if "algorithms" in details_dict:
- algorithms_data = details_dict.pop("algorithms")
- for category, alg_ids in algorithms_data.items():
- algorithms |= {category + x for x in alg_ids}
-
- return algorithms, FIPSCertificate.WebData(
- **{**details_dict, **vendor_dict, **related_files_dict, **validation_history_dict}
- )
-
- def _build_details_dict(self, details_div: Tag) -> dict[str, Any]:
- def parse_single_detail_entry(key, entry):
- normalized_key = DETAILS_KEY_NORMALIZATION_DICT[key]
- normalization_func = DETAILS_KEY_TO_NORMALIZATION_FUNCTION.get(normalized_key, None)
- normalized_entry = (
- FIPSHTMLParser.normalize_string(entry.text) if not normalization_func else normalization_func(entry)
- )
- return normalized_key, normalized_entry
-
- entries = details_div.find_all("div", class_="row padrow")
- entries = zip(
- [x.find("div", class_="col-md-3") for x in entries], [x.find("div", class_="col-md-9") for x in entries]
- )
- entries = [(FIPSHTMLParser.normalize_string(key.text), entry) for key, entry in entries]
- entries = [parse_single_detail_entry(*x) for x in entries if x[0] in DETAILS_KEY_NORMALIZATION_DICT.keys()]
- entries = {x: y for x, y in entries}
-
- if "caveat" in entries:
- entries["mentioned_certs"] = FIPSHTMLParser.get_mentioned_certs_from_caveat(entries["caveat"])
-
- # Temporarily disabled, as this isn't extracting anything useful. Only UNKNOWN#1-9 algs were extracted over whole dataset.
- # if "description" in entries:
- # algs = FIPSHTMLParser.get_algs_from_description(entries["description"])
- # if "algorithms" in entries:
- # entries["algorithms"].update({"UNKNOWN": x for x in algs})
- # else:
- # entries["algorithms"] = {"UNKNOWN": x for x in algs}
-
- return entries
-
- @staticmethod
- def _build_vendor_dict(vendor_div: Tag) -> dict[str, Any]:
- if not (link := vendor_div.find("a")):
- return {"vendor_url": None, "vendor": list(vendor_div.find("div", "panel-body").children)[0].strip()}
- else:
- return {"vendor_url": link.get("href"), "vendor": link.text.strip()}
-
- @staticmethod
- def _build_related_files_dict(related_files_div: Tag) -> dict[str, Any]:
- if cert_link := [x for x in related_files_div.find_all("a") if "Certificate" in x.text]:
- return {"certificate_pdf_url": constants.FIPS_BASE_URL + cert_link[0].get("href")}
- else:
- return {"certificate_pdf_url": None}
-
- @staticmethod
- def _build_validation_history_dict(validation_history_div: Tag) -> dict[str, Any]:
- def parse_row(row):
- validation_date, validation_type, lab = row.find_all("td")
- return FIPSCertificate.ValidationHistoryEntry(
- dateutil.parser.parse(validation_date.text).date(), validation_type.text, lab.text
- )
-
- rows = validation_history_div.find("tbody").find_all("tr")
- history: list[FIPSCertificate.ValidationHistoryEntry] | None = [parse_row(x) for x in rows] if rows else None
- return {"validation_history": history}
-
- @staticmethod
- def get_mentioned_certs_from_caveat(caveat: str) -> dict[str, int]:
- ids_found: dict[str, int] = {}
- r_key = r"(?P<word>\w+)?\s?(?:#\s?|Cert\.?(?!.\s)\s?|Certificate\s?)+(?P<id>\d+)"
- for m in re.finditer(r_key, caveat):
- if m.group("word") and m.group("word").lower() in {"rsa", "shs", "dsa", "pkcs", "aes"}:
- continue
- if m.group("id") in ids_found:
- ids_found[m.group("id")] += 1
- else:
- ids_found[m.group("id")] = 1
- return ids_found
-
- @staticmethod
- def get_algs_from_description(description: str) -> set[str]:
- return {m.group() for m in re.finditer(FIPS_ALGS_IN_TABLE, description)}
-
- @staticmethod
- def parse_algorithms(algorithms_div: Tag) -> dict[str, set[str]]:
- rows = algorithms_div.find("tbody").find_all("tr")
- dct: dict[str, set[str]] = dict()
- for row in rows:
- cells = row.find_all("td")
- dct[cells[0].text] = {m.group() for m in re.finditer(FIPS_ALGS_IN_TABLE, cells[1].text)}
- return dct
-
- @staticmethod
- def normalize_string(string: str) -> str:
- return " ".join(string.split())
-
- @staticmethod
- def parse_tested_configurations(tested_configurations: Tag) -> list[str] | None:
- configurations = [y.text for y in tested_configurations.find_all("li")]
- return configurations if not configurations == ["N/A"] else None
-
- @staticmethod
- def normalize_embodiment(embodiment_element: Tag) -> str:
- text = FIPSHTMLParser.normalize_string(embodiment_element.text)
- embodiment_normalization_dict = {
- "Multi-chip embedded": "Multi-Chip Embedded",
- "Multi-chip Standalone": "Multi-Chip Stand Alone",
- "Multi-chip standalone": "Multi-Chip Stand Alone",
- "Single-chip": "Single Chip",
- }
- return embodiment_normalization_dict.get(text, text)
-
-
-DETAILS_KEY_NORMALIZATION_DICT: Final[dict[str, str]] = {
- "Module Name": "module_name",
- "Standard": "standard",
- "Status": "status",
- "Sunset Date": "date_sunset",
- "Validation Dates": "date_validation",
- "Overall Level": "level",
- "Caveat": "caveat",
- "Security Level Exceptions": "exceptions",
- "Module Type": "module_type",
- "Embodiment": "embodiment",
- "Approved Algorithms": "algorithms",
- "Tested Configuration(s)": "tested_conf",
- "Description": "description",
- "Historical Reason": "historical_reason",
- "Hardware Versions": "hw_versions",
- "Firmware Versions": "fw_versions",
- "Revoked Reason": "revoked_reason",
- "Revoked Link": "revoked_link",
- "Software Versions": "sw_versions",
- "Product URL": "product_url",
-}
-
-DETAILS_KEY_TO_NORMALIZATION_FUNCTION: dict[str, Callable] = {
- "date_sunset": lambda x: dateutil.parser.parse(x.text).date(),
- "algorithms": getattr(FIPSHTMLParser, "parse_algorithms"),
- "tested_conf": getattr(FIPSHTMLParser, "parse_tested_configurations"),
- "exceptions": lambda x: [y.text for y in x.find_all("li")],
- "status": lambda x: FIPSHTMLParser.normalize_string(x.text).lower(),
- "level": lambda x: int(FIPSHTMLParser.normalize_string(x.text)),
- "embodiment": getattr(FIPSHTMLParser, "normalize_embodiment"),
-}
-
-
-class FIPSCertificate(
- Certificate["FIPSCertificate", "FIPSCertificate.Heuristics", "FIPSCertificate.PdfData"],
- PandasSerializableType,
- ComplexSerializableType,
-):
- """
- Data structure for common FIPS 140 certificate. Contains several inner classes that layer the data logic.
- Can be serialized into/from json (`ComplexSerializableType`).
- Is basic element of `FIPSDataset`. The functionality is mostly related to holding data and transformations that
- the certificate can handle itself. `FIPSDataset` class then instrument this functionality.
- """
-
- pandas_columns: ClassVar[list[str]] = [
- "dgst",
- "cert_id",
- "name",
- "status",
- "standard",
- "type",
- "level",
- "embodiment",
- "date_validation",
- "date_sunset",
- "algorithms",
- "extracted_versions",
- "cpe_matches",
- "verified_cpe_matches",
- "related_cves",
- "module_directly_referenced_by",
- "module_indirectly_referenced_by",
- "module_directly_referencing",
- "module_indirectly_referencing",
- "policy_directly_referenced_by",
- "policy_indirectly_referenced_by",
- "policy_directly_referencing",
- "policy_indirectly_referencing",
- ]
-
- @dataclass(eq=True)
- class InternalState(ComplexSerializableType):
- """
- Holds state of the `FIPSCertificate`
- """
-
- module_download_ok: bool
- policy_download_ok: bool
-
- policy_convert_garbage: bool
- policy_convert_ok: bool
-
- module_extract_ok: bool
- policy_extract_ok: bool
-
- policy_pdf_hash: str | None
- policy_txt_hash: str | None
-
- policy_pdf_path: Path
- policy_txt_path: Path
- module_html_path: Path
-
- def __init__(
- self,
- module_download_ok: bool = False,
- policy_download_ok: bool = False,
- policy_convert_garbage: bool = False,
- policy_convert_ok: bool = False,
- module_extract_ok: bool = False,
- policy_extract_ok: bool = False,
- policy_pdf_hash: str | None = None,
- policy_txt_hash: str | None = None,
- ):
- self.module_download_ok = module_download_ok
- self.policy_download_ok = policy_download_ok
- self.policy_convert_garbage = policy_convert_garbage
- self.policy_convert_ok = policy_convert_ok
- self.module_extract_ok = module_extract_ok
- self.policy_extract_ok = policy_extract_ok
- self.policy_pdf_hash = policy_pdf_hash
- self.policy_txt_hash = policy_txt_hash
-
- @property
- def serialized_attributes(self) -> list[str]:
- return [
- "module_download_ok",
- "policy_download_ok",
- "policy_convert_garbage",
- "policy_convert_ok",
- "module_extract_ok",
- "policy_extract_ok",
- "policy_pdf_hash",
- "policy_txt_hash",
- ]
-
- def module_is_ok_to_download(self, fresh: bool = True) -> bool:
- return True if fresh else not self.module_download_ok
-
- def policy_is_ok_to_download(self, fresh: bool = True) -> bool:
- return True if fresh else not self.policy_download_ok
-
- def policy_is_ok_to_convert(self, fresh: bool = True) -> bool:
- return self.policy_download_ok if fresh else self.policy_download_ok and not self.policy_convert_ok
-
- def module_is_ok_to_analyze(self, fresh: bool = True) -> bool:
- return (
- self.module_download_ok and self.module_extract_ok
- if fresh
- else self.module_download_ok and not self.module_extract_ok
- )
-
- def policy_is_ok_to_analyze(self, fresh: bool = True) -> bool:
- return (
- self.policy_convert_ok and self.policy_extract_ok
- if fresh
- else self.policy_convert_ok and not self.policy_extract_ok
- )
-
- def set_local_paths(self, policies_pdf_dir: Path, policies_txt_dir: Path, modules_html_dir: Path) -> None:
- self.state.policy_pdf_path = (policies_pdf_dir / str(self.dgst)).with_suffix(".pdf")
- self.state.policy_txt_path = (policies_txt_dir / str(self.dgst)).with_suffix(".txt")
- self.state.module_html_path = (modules_html_dir / str(self.dgst)).with_suffix(".html")
-
- @dataclass(eq=True)
- class ValidationHistoryEntry(ComplexSerializableType):
- date: date
- validation_type: Literal["initial", "update"]
- lab: str
-
- @classmethod
- def from_dict(cls, dct: dict) -> FIPSCertificate.ValidationHistoryEntry:
- new_dct = dct.copy()
- new_dct["date"] = dateutil.parser.parse(dct["date"]).date()
- return cls(**new_dct)
-
- @dataclass(eq=True)
- class WebData(ComplexSerializableType):
- """
- Data structure for data obtained from scanning certificate webpage at NIST.gov
- """
-
- module_name: str | None = field(default=None)
- validation_history: list[FIPSCertificate.ValidationHistoryEntry] | None = field(default=None)
- vendor_url: str | None = field(default=None)
- vendor: str | None = field(default=None)
- certificate_pdf_url: str | None = field(default=None)
- module_type: str | None = field(default=None)
- standard: str | None = field(default=None)
- status: Literal["active", "historical", "revoked"] | None = field(default=None)
- level: Literal[1, 2, 3, 4] | None = field(default=None)
- caveat: str | None = field(default=None)
- exceptions: list[str] | None = field(default=None)
- embodiment: str | None = field(default=None)
- description: str | None = field(default=None)
- tested_conf: list[str] | None = field(default=None)
- hw_versions: str | None = field(default=None)
- fw_versions: str | None = field(default=None)
- sw_versions: str | None = field(default=None)
- mentioned_certs: dict[str, int] | None = field(default=None) # Cert_id: n_occurences
- historical_reason: str | None = field(default=None)
- date_sunset: date | None = field(default=None)
- revoked_reason: str | None = field(default=None)
- revoked_link: str | None = field(default=None)
-
- # Those below are left unused at the moment
- # product_url: Optional[str] = field(default=None)
-
- def __repr__(self) -> str:
- return (
- self.module_name
- if self.module_name is not None
- else "" + " created by " + self.vendor
- if self.vendor is not None
- else ""
- )
-
- def __str__(self) -> str:
- return repr(self)
-
- @classmethod
- def from_dict(cls, dct: dict) -> FIPSCertificate.WebData:
- new_dct = dct.copy()
- if new_dct["date_sunset"]:
- new_dct["date_sunset"] = dateutil.parser.parse(new_dct["date_sunset"]).date()
- return cls(**dct)
-
- @dataclass(eq=True)
- class PdfData(BasePdfData, ComplexSerializableType):
- """
- Data structure that holds data obtained from scanning pdf files (or their converted txt documents).
- """
-
- keywords: dict = field(default_factory=dict)
- policy_metadata: dict[str, Any] = field(default_factory=dict)
-
- @property
- def certlike_algorithm_numbers(self) -> set[str]:
- """Returns numbers of certificates from keywords["fips_certlike"]["Certlike"]"""
- if self.keywords and "fips_certlike" in self.keywords:
- fips_certlike = self.keywords["fips_certlike"].get("Certlike", dict())
- matches = {re.search(r"#\s{0,1}\d{1,4}", x) for x in fips_certlike.keys()}
- return {"".join([x for x in match.group() if x.isdigit()]) for match in matches if match}
- else:
- return set()
-
- @dataclass(eq=True)
- class Heuristics(BaseHeuristics, ComplexSerializableType):
- """
- Data structure that holds data obtained by processing the certificate and applying various heuristics.
- """
-
- algorithms: set[str] = field(default_factory=set)
- extracted_versions: set[str] = field(default_factory=set)
- cpe_matches: set[str] | None = field(default=None)
- verified_cpe_matches: set[CPE] | None = field(default=None)
- related_cves: set[str] | None = field(default=None)
- policy_prunned_references: set[str] = field(default_factory=set)
- module_prunned_references: set[str] = field(default_factory=set)
- policy_processed_references: References = field(default_factory=References)
- module_processed_references: References = field(default_factory=References)
- direct_transitive_cves: set[str] | None = field(default=None)
- indirect_transitive_cves: set[str] | None = field(default=None)
-
- @property
- def algorithm_numbers(self) -> set[str]:
- """Returns numbers of algorithms"""
-
- def alg_to_number(alg: str) -> str:
- return "".join([x for x in alg.split("#")[1] if x.isdigit()])
-
- return {alg_to_number(x) for x in self.algorithms}
-
- @property
- def dgst(self) -> str:
- """
- Returns primary key of the certificate, its id.
- """
- return fips_dgst(self.cert_id)
-
- @property
- def manufacturer(self) -> str | None: # type: ignore
- return self.web_data.vendor
-
- @property
- def module_html_url(self) -> str:
- return constants.FIPS_MODULE_URL.format(self.cert_id)
-
- @property
- def policy_pdf_url(self) -> str:
- return constants.FIPS_SP_URL.format(self.cert_id)
-
- @property
- def name(self) -> str | None: # type: ignore
- return self.web_data.module_name
-
- @property
- def label_studio_title(self) -> str:
- return (
- "Vendor: "
- + str(self.web_data.vendor)
- + "\n"
- + "Module name: "
- + str(self.web_data.module_name)
- + "\n"
- + "HW version: "
- + str(self.web_data.hw_versions)
- + "\n"
- + "FW version: "
- + str(self.web_data.fw_versions)
- )
-
- def __init__(
- self,
- cert_id: str,
- web_data: FIPSCertificate.WebData | None = None,
- pdf_data: FIPSCertificate.PdfData | None = None,
- heuristics: FIPSCertificate.Heuristics | None = None,
- state: InternalState | None = None,
- ):
- super().__init__()
-
- self.cert_id = cert_id
- self.web_data: FIPSCertificate.WebData = web_data if web_data else FIPSCertificate.WebData()
- self.pdf_data: FIPSCertificate.PdfData = pdf_data if pdf_data else FIPSCertificate.PdfData()
- self.heuristics: FIPSCertificate.Heuristics = heuristics if heuristics else FIPSCertificate.Heuristics()
- self.state: FIPSCertificate.InternalState = state if state else FIPSCertificate.InternalState()
-
- @property
- def pandas_tuple(self) -> tuple:
- return (
- self.dgst,
- self.cert_id,
- self.web_data.module_name,
- self.web_data.status,
- self.web_data.standard,
- self.web_data.module_type,
- self.web_data.level,
- self.web_data.embodiment,
- self.web_data.validation_history[0].date if self.web_data.validation_history else np.nan,
- self.web_data.date_sunset,
- self.heuristics.algorithms,
- self.heuristics.extracted_versions,
- self.heuristics.cpe_matches,
- self.heuristics.verified_cpe_matches,
- self.heuristics.related_cves,
- self.heuristics.module_processed_references.directly_referenced_by,
- self.heuristics.module_processed_references.indirectly_referenced_by,
- self.heuristics.module_processed_references.directly_referencing,
- self.heuristics.module_processed_references.indirectly_referencing,
- self.heuristics.policy_processed_references.directly_referenced_by,
- self.heuristics.policy_processed_references.indirectly_referenced_by,
- self.heuristics.policy_processed_references.directly_referencing,
- self.heuristics.policy_processed_references.indirectly_referencing,
- )
-
- @staticmethod
- def parse_html_module(cert: FIPSCertificate) -> FIPSCertificate:
- with cert.state.module_html_path.open("r") as handle:
- soup = BeautifulSoup(handle, "html5lib")
-
- parser = FIPSHTMLParser(soup)
- algorithms, cert.web_data = parser.get_web_data_and_algorithms()
- cert.heuristics.algorithms |= algorithms
- cert.state.module_extract_ok = True
-
- return cert
-
- @staticmethod
- def download_module(cert: FIPSCertificate) -> FIPSCertificate:
- if (exit_code := helpers.download_file(cert.module_html_url, cert.state.module_html_path)) != requests.codes.ok:
- error_msg = f"failed to download html module from {cert.module_html_url}, code {exit_code}"
- logger.error(f"Cert dgst: {cert.dgst} " + error_msg)
- cert.state.module_download_ok = False
- else:
- cert.state.module_download_ok = True
- return cert
-
- @staticmethod
- def download_policy(cert: FIPSCertificate) -> FIPSCertificate:
- if (exit_code := helpers.download_file(cert.policy_pdf_url, cert.state.policy_pdf_path)) != requests.codes.ok:
- error_msg = f"failed to download pdf policy from {cert.policy_pdf_url}, code {exit_code}"
- logger.error(f"Cert dgst: {cert.dgst} " + error_msg)
- cert.state.policy_download_ok = False
- else:
- cert.state.policy_download_ok = True
- cert.state.policy_pdf_hash = helpers.get_sha256_filepath(cert.state.policy_pdf_path)
- return cert
-
- @staticmethod
- def convert_policy_pdf(cert: FIPSCertificate) -> FIPSCertificate:
- """
- Converts policy pdf -> txt
- """
- ocr_done, ok_result = sec_certs.utils.pdf.convert_pdf_file(
- cert.state.policy_pdf_path, cert.state.policy_txt_path
- )
-
- # If OCR was done and the result was garbage
- cert.state.policy_convert_garbage = ocr_done
- # And put the whole result into convert_ok
- cert.state.policy_convert_ok = ok_result
-
- if not ok_result:
- error_msg = "Failed to convert policy pdf->txt"
- logger.error(f"Cert dgst: {cert.dgst}" + error_msg)
- else:
- cert.state.policy_txt_hash = helpers.get_sha256_filepath(cert.state.policy_txt_path)
-
- return cert
-
- @staticmethod
- def extract_policy_pdf_metadata(cert: FIPSCertificate) -> FIPSCertificate:
- """
- Extract the PDF metadata from the security policy.
- """
- _, metadata = sec_certs.utils.pdf.extract_pdf_metadata(cert.state.policy_pdf_path)
-
- if metadata:
- cert.pdf_data.policy_metadata = metadata
- else:
- cert.pdf_data.policy_metadata = dict()
- cert.state.policy_extract_ok = False
- return cert
-
- @staticmethod
- def extract_policy_pdf_keywords(cert: FIPSCertificate) -> FIPSCertificate:
- """
- Extract keywords from policy document
- """
- keywords = sec_certs.utils.extract.extract_keywords(cert.state.policy_txt_path, fips_rules)
- if not keywords:
- cert.state.policy_extract_ok = False
- else:
- cert.pdf_data.keywords = keywords
- return cert
-
- @staticmethod
- def get_algorithms_from_policy_tables(cert: FIPSCertificate):
- """
- Retrieves IDs of algorithms from tables inside security policy pdfs.
- External library is used to handle this.
- """
- if table_rich_page_numbers := tables.find_pages_with_tables(cert.state.policy_txt_path):
- pdf.repair_pdf(cert.state.policy_pdf_path)
- try:
- tabular_data = read_pdf(cert.state.policy_pdf_path, pages=list(table_rich_page_numbers), silent=True)
- cert.heuristics.algorithms |= set(
- itertools.chain.from_iterable(tables.get_algs_from_table(df.to_string()) for df in tabular_data)
- )
- except Exception as e:
- logger.warning(f"Error when parsing tables from {cert.dgst}: {e}")
- cert.state.policy_extract_ok = False
-
- def prune_referenced_cert_ids(self) -> None:
- """
- This method goes through all IDs (numbers) that correspond to FIPS Certificates and are stored in
- pdf_data.keywords or web_data.mentioned_certs. It performs prunning of these attributes and fills attributes
- heuristics.prunned_module_references and heuristics.prunned_policy_references. These variables are further
- processed and Reference objects are created from them.
- """
- html_module_ids = set(self.web_data.mentioned_certs.keys()) if self.web_data.mentioned_certs else set()
- self.heuristics.module_prunned_references = self._prune_reference_ids_variable(html_module_ids)
-
- if self.pdf_data.keywords:
- pdf_policy_ids = set(self.pdf_data.keywords["fips_cert_id"].get("Cert", dict()).keys())
- pdf_policy_ids = {"".join([y for y in x if y.isdigit()]) for x in pdf_policy_ids}
- else:
- pdf_policy_ids = set()
-
- self.heuristics.policy_prunned_references = self._prune_reference_ids_variable(pdf_policy_ids)
-
- def compute_heuristics_version(self) -> None:
- """
- Heuristically computes the version of the product.
- """
- versions_for_extraction = ""
- if self.web_data.module_name:
- versions_for_extraction += f" {self.web_data.module_name}"
- if self.web_data.hw_versions:
- versions_for_extraction += f" {self.web_data.hw_versions}"
- if self.web_data.fw_versions:
- versions_for_extraction += f" {self.web_data.fw_versions}"
- self.heuristics.extracted_versions = helpers.compute_heuristics_version(versions_for_extraction)
-
- def _prune_reference_ids_variable(self, attribute_to_prune: set[str]) -> set[str]:
- """
- Prunnes cert_ids from variable "attribute_to_prune", return result. Steps:
- 0. Consider only ids != self.cert_id
- 1. Consider only ids > config.always_false_positive_fips_cert_id_threshold
- 2. Consider only ids s.t. they don't appear in self.heuristics.algorithms
- 3. Consider only ids s.t. they don't appear in self.pdf_data.keywords["fips_certlike"]["Certlike"]
- """
- prunned = {x for x in attribute_to_prune if x != self.cert_id}
- prunned = {x for x in prunned if int(x) > config.always_false_positive_fips_cert_id_threshold}
- prunned = {x for x in prunned if x not in self.heuristics.algorithm_numbers}
- prunned = {x for x in prunned if x not in self.pdf_data.certlike_algorithm_numbers}
-
- return prunned
diff --git a/sec_certs/sample/fips_algorithm.py b/sec_certs/sample/fips_algorithm.py
deleted file mode 100644
index 16f19e64..00000000
--- a/sec_certs/sample/fips_algorithm.py
+++ /dev/null
@@ -1,50 +0,0 @@
-from __future__ import annotations
-
-from dataclasses import dataclass
-from datetime import date
-from typing import ClassVar
-
-from sec_certs import constants
-from sec_certs.serialization.json import ComplexSerializableType
-from sec_certs.serialization.pandas import PandasSerializableType
-
-
-@dataclass(eq=True, frozen=True)
-class FIPSAlgorithm(PandasSerializableType, ComplexSerializableType):
- """
- Data structure for algorithm of `FIPSCertificate`
- """
-
- alg_number: str
- algorithm_type: str
- vendor: str
- implementation_name: str
- validation_date: date
-
- pandas_columns: ClassVar[list[str]] = [
- "dgst",
- "alg_number",
- "algorithm_type",
- "vendor",
- "implementation_name",
- "validation_date",
- ]
-
- @property
- def pandas_tuple(self) -> tuple:
- return (
- self.dgst,
- self.alg_number,
- self.algorithm_type,
- self.vendor,
- self.implementation_name,
- self.validation_date,
- )
-
- @property
- def dgst(self) -> str:
- return f"{self.algorithm_type}{self.alg_number}"
-
- @property
- def page_url(self) -> str:
- return constants.FIPS_ALG_URL.format(self.algorithm_type, self.alg_number)
diff --git a/sec_certs/sample/fips_iut.py b/sec_certs/sample/fips_iut.py
deleted file mode 100644
index cb521ee4..00000000
--- a/sec_certs/sample/fips_iut.py
+++ /dev/null
@@ -1,166 +0,0 @@
-from __future__ import annotations
-
-from dataclasses import dataclass
-from datetime import date, datetime
-from pathlib import Path
-from tempfile import NamedTemporaryFile
-from typing import Iterator, Mapping
-
-import requests
-from bs4 import BeautifulSoup, Tag
-
-from sec_certs import constants
-from sec_certs.config.configuration import config
-from sec_certs.serialization.json import ComplexSerializableType
-from sec_certs.utils.helpers import to_utc
-
-
-@dataclass(frozen=True)
-class IUTEntry(ComplexSerializableType):
- module_name: str
- vendor_name: str
- standard: str
- iut_date: date
-
- def to_dict(self) -> dict[str, str]:
- return {**self.__dict__, "iut_date": self.iut_date.isoformat()}
-
- @classmethod
- def from_dict(cls, dct: Mapping) -> IUTEntry:
- return cls(
- dct["module_name"],
- dct["vendor_name"],
- dct["standard"],
- date.fromisoformat(dct["iut_date"]),
- )
-
-
-@dataclass
-class IUTSnapshot(ComplexSerializableType):
- entries: set[IUTEntry]
- timestamp: datetime
- last_updated: date
- displayed: int | None
- not_displayed: int | None
- total: int | None
-
- def __len__(self) -> int:
- return len(self.entries)
-
- def __iter__(self) -> Iterator[IUTEntry]:
- yield from self.entries
-
- def to_dict(self) -> dict[str, int | None | list[IUTEntry] | str]:
- return {
- "entries": list(self.entries),
- "timestamp": self.timestamp.isoformat(),
- "last_updated": self.last_updated.isoformat(),
- "displayed": self.displayed,
- "not_displayed": self.not_displayed,
- "total": self.total,
- }
-
- @classmethod
- def from_dict(cls, dct: Mapping) -> IUTSnapshot:
- return cls(
- set(dct["entries"]),
- datetime.fromisoformat(dct["timestamp"]),
- date.fromisoformat(dct["last_updated"]),
- dct["displayed"],
- dct["not_displayed"],
- dct["total"],
- )
-
- @classmethod
- def from_page(cls, content: bytes, snapshot_date: datetime) -> IUTSnapshot:
- """
- Get an IUT snapshot from a HTML dump of the FIPS website.
- """
- if not content:
- raise ValueError("Empty content in IUT.")
- soup = BeautifulSoup(content, "html5lib")
- tables = soup.find_all("table")
- if len(tables) != 1:
- raise ValueError("Not only a single table in IUT.")
-
- last_updated_elem = next(
- filter(
- lambda e: isinstance(e, Tag) and e.name == "p",
- soup.find(id="content").next_siblings,
- )
- )
- last_updated_text = str(last_updated_elem.string).strip()
- last_updated = datetime.strptime(last_updated_text, "Last Updated: %m/%d/%Y").date()
- table = tables[0].find("tbody")
- lines = table.find_all("tr")
- entries = {
- IUTEntry(
- str(line[0].string),
- str(line[1].string),
- str(line[2].string),
- datetime.strptime(str(line[3].string), "%m/%d/%Y").date(),
- )
- for line in map(lambda tr: tr.find_all("td"), lines)
- }
-
- # Parse footer
- footer = soup.find(id="IUTFooter")
- displayed: int | None
- not_displayed: int | None
- total: int | None
-
- if footer:
- footer_lines = footer.find_all("tr")
- displayed = int(footer_lines[0].find_all("td")[1].text)
- not_displayed = int(footer_lines[1].find_all("td")[1].text)
- total = int(footer_lines[2].find_all("td")[1].text)
- else:
- displayed, not_displayed, total = (None, None, None)
-
- return cls(
- entries=entries,
- timestamp=snapshot_date,
- last_updated=last_updated,
- displayed=displayed,
- not_displayed=not_displayed,
- total=total,
- )
-
- @classmethod
- def from_dump(cls, dump_path: str | Path, snapshot_date: datetime | None = None) -> IUTSnapshot:
- """
- Get an IUT snapshot from a HTML file dump of the FIPS website.
- """
- dump_path = Path(dump_path)
- if snapshot_date is None:
- try:
- snapshot_date = to_utc(datetime.fromisoformat(dump_path.name[len("fips_iut_") : -len(".html")]))
- except Exception:
- raise ValueError("snapshot_date not given and could not be inferred from filename.")
- with dump_path.open("rb") as f:
- content = f.read()
- return cls.from_page(content, snapshot_date)
-
- @classmethod
- def from_web(cls) -> IUTSnapshot:
- """
- Get an IUT snapshot from the FIPS website right now.
- """
- iut_resp = requests.get(constants.FIPS_IUT_URL)
- if iut_resp.status_code != 200:
- raise ValueError(f"Getting IUT snapshot failed: {iut_resp.status_code}")
-
- snapshot_date = to_utc(datetime.now())
- return cls.from_page(iut_resp.content, snapshot_date)
-
- @classmethod
- def from_web_latest(cls) -> IUTSnapshot:
- """
- Get a IUT snapshot from seccerts.org.
- """
- iut_resp = requests.get(config.fips_iut_latest_snapshot)
- if iut_resp.status_code != 200:
- raise ValueError(f"Getting MIP snapshot failed: {iut_resp.status_code}")
- with NamedTemporaryFile() as tmpfile:
- tmpfile.write(iut_resp.content)
- return cls.from_json(tmpfile.name)
diff --git a/sec_certs/sample/fips_mip.py b/sec_certs/sample/fips_mip.py
deleted file mode 100644
index 6918d2aa..00000000
--- a/sec_certs/sample/fips_mip.py
+++ /dev/null
@@ -1,242 +0,0 @@
-from __future__ import annotations
-
-import logging
-from dataclasses import dataclass
-from datetime import date, datetime
-from enum import Enum
-from pathlib import Path
-from tempfile import NamedTemporaryFile
-from typing import Iterator, Mapping
-
-import requests
-from bs4 import BeautifulSoup, Tag
-
-from sec_certs import constants
-from sec_certs.config.configuration import config
-from sec_certs.constants import FIPS_MIP_STATUS_RE
-from sec_certs.serialization.json import ComplexSerializableType
-from sec_certs.utils.helpers import to_utc
-
-logger = logging.getLogger(__name__)
-
-
-class MIPStatus(Enum):
- IN_REVIEW = "In Review"
- REVIEW_PENDING = "Review Pending"
- COORDINATION = "Coordination"
- FINALIZATION = "Finalization"
-
-
-@dataclass(frozen=True)
-class MIPEntry(ComplexSerializableType):
- module_name: str
- vendor_name: str
- standard: str
- status: MIPStatus | None
- status_since: date | None
-
- def to_dict(self) -> dict[str, str | MIPStatus | None | date | None]:
- return {
- **self.__dict__,
- "status": self.status.value if self.status else None,
- "status_since": self.status_since.isoformat() if self.status_since else None,
- }
-
- @classmethod
- def from_dict(cls, dct: Mapping) -> MIPEntry:
- return cls(
- dct["module_name"],
- dct["vendor_name"],
- dct["standard"],
- MIPStatus(dct["status"]) if dct["status"] else None,
- date.fromisoformat(dct["status_since"]) if dct.get("status_since") else None,
- )
-
-
-@dataclass
-class MIPSnapshot(ComplexSerializableType):
- entries: set[MIPEntry]
- timestamp: datetime
- last_updated: date
- displayed: int
- not_displayed: int
- total: int
-
- def __len__(self) -> int:
- return len(self.entries)
-
- def __iter__(self) -> Iterator[MIPEntry]:
- yield from self.entries
-
- def to_dict(self) -> dict[str, int | str | list[MIPEntry]]:
- return {
- "entries": list(self.entries),
- "timestamp": self.timestamp.isoformat(),
- "last_updated": self.last_updated.isoformat(),
- "displayed": self.displayed,
- "not_displayed": self.not_displayed,
- "total": self.total,
- }
-
- @classmethod
- def from_dict(cls, dct: Mapping) -> MIPSnapshot:
- return cls(
- set(dct["entries"]),
- datetime.fromisoformat(dct["timestamp"]),
- date.fromisoformat(dct["last_updated"]),
- dct["displayed"],
- dct["not_displayed"],
- dct["total"],
- )
-
- @classmethod
- def _extract_entries_1(cls, lines):
- """Works until 2020.10.28 (including)."""
- entries = set()
- for tr in lines:
- tds = tr.find_all("td")
- status = None
- if "mip-highlight" in tds[-1]["class"]:
- status = MIPStatus.FINALIZATION
- elif "mip-highlight" in tds[-2]["class"]:
- status = MIPStatus.COORDINATION
- elif "mip-highlight" in tds[-3]["class"]:
- status = MIPStatus.REVIEW_PENDING
- elif "mip-highlight" in tds[-4]["class"]:
- status = MIPStatus.IN_REVIEW
- entries.add(MIPEntry(str(tds[0].string), str(tds[1].string), str(tds[2].string), status, None))
- return entries
-
- @classmethod
- def _extract_entries_2(cls, lines):
- """Works until 2021.04.20 (including)."""
- return {
- MIPEntry(
- str(line[0].string), str(line[1].string), str(line[2].string), MIPStatus(str(line[3].string)), None
- )
- for line in map(lambda tr: tr.find_all("td"), lines)
- }
-
- @classmethod
- def _extract_entries_3(cls, lines):
- """Works until 2022.03.23 (including)."""
- return {
- MIPEntry(
- str(line[0].string),
- str(" ".join(line[1].find_all(text=True, recursive=False)).strip()),
- str(line[2].string),
- MIPStatus(str(line[3].string)),
- None,
- )
- for line in map(lambda tr: tr.find_all("td"), lines)
- }
-
- @classmethod
- def _extract_entries_4(cls, lines):
- """Works now."""
- entries = set()
- for line in map(lambda tr: tr.find_all("td"), lines):
- module_name = str(line[0].string)
- vendor_name = str(" ".join(line[1].find_all(text=True, recursive=False)).strip())
- standard = str(line[2].string)
- status_line = FIPS_MIP_STATUS_RE.match(str(line[3].string))
- if status_line is None:
- raise ValueError("Cannot parse MIP status line.")
- status = MIPStatus(status_line.group("status"))
- since = datetime.strptime(status_line.group("since"), "%m/%d/%Y").date()
- entries.add(MIPEntry(module_name, vendor_name, standard, status, since))
- return entries
-
- @classmethod
- def _extract_entries(cls, lines, snapshot_date):
- if snapshot_date <= datetime(2020, 10, 28):
- entries = cls._extract_entries_1(lines)
- elif snapshot_date <= datetime(2021, 4, 20):
- entries = cls._extract_entries_2(lines)
- elif snapshot_date <= datetime(2022, 3, 23):
- entries = cls._extract_entries_3(lines)
- else:
- entries = cls._extract_entries_4(lines)
- return entries
-
- @classmethod
- def from_page(cls, content: bytes, snapshot_date: datetime) -> MIPSnapshot:
- """
- Get a MIP snapshot from a HTML dump of the FIPS website.
- """
- if not content:
- raise ValueError("Empty content in MIP.")
- soup = BeautifulSoup(content, "html5lib")
- tables = soup.find_all("table")
- if len(tables) != 1:
- raise ValueError("Not only a single table in MIP data.")
-
- # Parse Last Updated
- last_updated_elem = next(
- filter(
- lambda e: isinstance(e, Tag) and e.name == "p",
- soup.find(id="content").next_siblings,
- )
- )
- last_updated_text = str(last_updated_elem.string).strip()
- last_updated = datetime.strptime(last_updated_text, "Last Updated: %m/%d/%Y").date()
-
- # Parse entries
- table = tables[0].find("tbody")
- lines = table.find_all("tr")
- entries = cls._extract_entries(lines, snapshot_date)
-
- # Parse footer
- footer = soup.find(id="MIPFooter")
- footer_lines = footer.find_all("tr")
- displayed = int(footer_lines[0].find_all("td")[1].text)
- not_displayed = int(footer_lines[1].find_all("td")[1].text)
- total = int(footer_lines[2].find_all("td")[1].text)
-
- return cls(
- entries=entries,
- timestamp=snapshot_date,
- last_updated=last_updated,
- displayed=displayed,
- not_displayed=not_displayed,
- total=total,
- )
-
- @classmethod
- def from_dump(cls, dump_path: str | Path, snapshot_date: datetime | None = None) -> MIPSnapshot:
- """
- Get a MIP snapshot from a HTML file dump of the FIPS website.
- """
- dump_path = Path(dump_path)
- if snapshot_date is None:
- try:
- snapshot_date = to_utc(datetime.fromisoformat(dump_path.name[len("fips_mip_") : -len(".html")]))
- except Exception:
- raise ValueError("snapshot_date not given and could not be inferred from filename.")
- with dump_path.open("rb") as f:
- content = f.read()
- return cls.from_page(content, snapshot_date)
-
- @classmethod
- def from_web(cls) -> MIPSnapshot:
- """
- Get a MIP snapshot from the FIPS website right now.
- """
- mip_resp = requests.get(constants.FIPS_MIP_URL)
- if mip_resp.status_code != 200:
- raise ValueError(f"Getting MIP snapshot failed: {mip_resp.status_code}")
-
- snapshot_date = to_utc(datetime.now())
- return cls.from_page(mip_resp.content, snapshot_date)
-
- @classmethod
- def from_web_latest(cls) -> MIPSnapshot:
- """
- Get a MIP snapshot from seccerts.org.
- """
- mip_resp = requests.get(config.fips_mip_latest_snapshot)
- if mip_resp.status_code != 200:
- raise ValueError(f"Getting MIP snapshot failed: {mip_resp.status_code}")
- with NamedTemporaryFile() as tmpfile:
- tmpfile.write(mip_resp.content)
- return cls.from_json(tmpfile.name)
diff --git a/sec_certs/sample/protection_profile.py b/sec_certs/sample/protection_profile.py
deleted file mode 100644
index b7c2ec34..00000000
--- a/sec_certs/sample/protection_profile.py
+++ /dev/null
@@ -1,55 +0,0 @@
-from __future__ import annotations
-
-import copy
-import logging
-from dataclasses import dataclass
-from typing import Any
-
-import sec_certs.utils.sanitization as sanitization
-from sec_certs.serialization.json import ComplexSerializableType
-
-logger = logging.getLogger(__name__)
-
-
-@dataclass(frozen=True)
-class ProtectionProfile(ComplexSerializableType):
- """
- Object for holding protection profiles.
- """
-
- pp_name: str
- pp_eal: str | None
- pp_link: str | None = None
- pp_ids: frozenset[str] | None = None
-
- def __post_init__(self):
- super().__setattr__("pp_name", sanitization.sanitize_string(self.pp_name))
- super().__setattr__("pp_link", sanitization.sanitize_link(self.pp_link))
-
- @classmethod
- def from_dict(cls, dct: dict[str, Any]) -> ProtectionProfile:
- new_dct = copy.deepcopy(dct)
- new_dct["pp_ids"] = frozenset(new_dct["pp_ids"]) if new_dct["pp_ids"] else None
- return cls(*tuple(new_dct.values()))
-
- @classmethod
- def from_old_api_dict(cls, dct: dict[str, Any]) -> ProtectionProfile:
- pp_name = sanitization.sanitize_string(dct["csv_scan"]["cc_pp_name"])
- pp_link = sanitization.sanitize_link(dct["csv_scan"]["link_pp_document"])
- pp_ids = frozenset(dct["processed"]["cc_pp_csvid"]) if dct["processed"]["cc_pp_csvid"] else None
- eal_set = sanitization.sanitize_security_levels(dct["csv_scan"]["cc_security_level"])
-
- if not len(eal_set) <= 1:
- raise ValueError("EAL field should have single value or should be empty.")
-
- eal_str = list(eal_set)[0] if eal_set else None
-
- return cls(pp_name, eal_str, pp_link, pp_ids)
-
- def __eq__(self, other: object) -> bool:
- if not isinstance(other, ProtectionProfile):
- return False
- return self.pp_name == other.pp_name and self.pp_link == other.pp_link
-
- def __lt__(self, other: ProtectionProfile) -> bool:
- return self.pp_name < other.pp_name
diff --git a/sec_certs/sample/sar.py b/sec_certs/sample/sar.py
deleted file mode 100644
index 31359299..00000000
--- a/sec_certs/sample/sar.py
+++ /dev/null
@@ -1,59 +0,0 @@
-from __future__ import annotations
-
-import re
-from dataclasses import dataclass
-from typing import Any
-
-from sec_certs.serialization.json import ComplexSerializableType
-
-SAR_CLASS_MAPPING = {
- "APE": "Protection Profile evaluation",
- "ACE": "Protection Profile configuration evaluation",
- "ASE": "Security Target evaluation",
- "ADV": "Development",
- "AGD": "Guidance documents",
- "ALC": "Life-cycle support",
- "ATE": "Tests",
- "AVA": "Vulnerability assessment",
- "ACO": "Comoposition",
-}
-
-SAR_CLASSES = {x for x in SAR_CLASS_MAPPING}
-SAR_DICT_KEY = "cc_sar"
-
-
-@dataclass(frozen=True, eq=True)
-class SAR(ComplexSerializableType):
- family: str
- level: int
-
- @property
- def assurance_class(self):
- return SAR_CLASS_MAPPING.get(self.family.split("_")[0], None)
-
- @classmethod
- def from_string(cls, string: str) -> SAR:
- if not cls.contains_level(string):
- raise ValueError("SAR misses level integer")
- if not cls.matches_re(string):
- raise ValueError("SAR does not match any regular expression")
- family = string.split(".")[0]
- level = int(string.split(".")[1])
- return cls(family, level)
-
- @staticmethod
- def contains_level(string: str) -> bool:
- if len(string.split(".")) == 1:
- return False
- return True
-
- @staticmethod
- def matches_re(string: str) -> bool:
- return any(
- [re.match(sar_class + "(?:_[A-Z]{3,4}){1,2}(?:\\.[0-9]){0,2}", string) for sar_class in SAR_CLASS_MAPPING]
- )
-
- def __lt__(self, other: Any) -> bool:
- if not isinstance(other, SAR):
- raise ValueError(f"cannot compare {type(other)} with SAR.")
- return str(self) < str(other)