diff options
| author | J08nY | 2022-07-16 16:14:38 +0200 |
|---|---|---|
| committer | J08nY | 2022-07-16 16:14:38 +0200 |
| commit | e0a2a1420cf3c21f0522b4de6db28446b19bbb69 (patch) | |
| tree | 219604f10de4ca97c818290d01cb7ce870778819 | |
| parent | ef2a3defe1138dc15ba959df5a631ed16894df00 (diff) | |
| download | sec-certs-e0a2a1420cf3c21f0522b4de6db28446b19bbb69.tar.gz sec-certs-e0a2a1420cf3c21f0522b4de6db28446b19bbb69.tar.zst sec-certs-e0a2a1420cf3c21f0522b4de6db28446b19bbb69.zip | |
Use html5lib instead of html.parser.
| -rw-r--r-- | sec_certs/dataset/common_criteria.py | 6 | ||||
| -rw-r--r-- | sec_certs/dataset/fips.py | 2 | ||||
| -rw-r--r-- | sec_certs/dataset/fips_algorithm.py | 4 | ||||
| -rw-r--r-- | sec_certs/sample/fips.py | 2 | ||||
| -rw-r--r-- | sec_certs/sample/fips_iut.py | 2 | ||||
| -rw-r--r-- | sec_certs/sample/fips_mip.py | 2 |
6 files changed, 9 insertions, 9 deletions
diff --git a/sec_certs/dataset/common_criteria.py b/sec_certs/dataset/common_criteria.py index 378bca61..d53f1e35 100644 --- a/sec_certs/dataset/common_criteria.py +++ b/sec_certs/dataset/common_criteria.py @@ -603,7 +603,7 @@ class CCDataset(Dataset[CommonCriteriaCert], ComplexSerializableType): cat_dict = {x: y for (x, y) in zip(cc_table_ids, cc_categories)} with file.open("r") as handle: - soup = BeautifulSoup(handle, "html.parser") + soup = BeautifulSoup(handle, "html5lib") certs = {} for key, val in cat_dict.items(): @@ -1006,7 +1006,7 @@ class CCSchemeDataset: resp = conn.get(url, headers={"User-Agent": "seccerts.org"}) if resp.status_code != requests.codes.ok: raise ValueError(f"Unable to download: status={resp.status_code}") - return BeautifulSoup(resp.content, "html.parser") + return BeautifulSoup(resp.content, "html5lib") @staticmethod def get_australia_in_evaluation(): @@ -1500,7 +1500,7 @@ class CCSchemeDataset: page = pages.pop() csrf = soup.find("form", id="fm").find("input", attrs={"name": "csrf"})["value"] resp = session.post(url, data={"csrf": csrf, "selectPage": page, "product_class": product_class}) - soup = BeautifulSoup(resp.content, "html.parser") + soup = BeautifulSoup(resp.content, "html5lib") tbody = soup.find("table", class_="cpl").find("tbody") for tr in tbody.find_all("tr"): tds = tr.find_all("td") diff --git a/sec_certs/dataset/fips.py b/sec_certs/dataset/fips.py index 64fd636e..1cb67480 100644 --- a/sec_certs/dataset/fips.py +++ b/sec_certs/dataset/fips.py @@ -164,7 +164,7 @@ class FIPSDataset(Dataset[FIPSCertificate], ComplexSerializableType): def _get_certificates_from_html(self, html_file: Path, update: bool = False) -> Set[str]: logger.info(f"Getting certificate ids from {html_file}") with open(html_file, "r", encoding="utf-8") as handle: - html = BeautifulSoup(handle.read(), "html.parser") + html = BeautifulSoup(handle.read(), "html5lib") table = [x for x in html.find(id="searchResultsTable").tbody.contents if x != "\n"] entries: Set[str] = set() diff --git a/sec_certs/dataset/fips_algorithm.py b/sec_certs/dataset/fips_algorithm.py index caaa5a7b..e9e3680c 100644 --- a/sec_certs/dataset/fips_algorithm.py +++ b/sec_certs/dataset/fips_algorithm.py @@ -40,7 +40,7 @@ class FIPSAlgorithmDataset(Dataset, ComplexSerializableType): logger.error("Couldn't download first page of algo dataset") with open(self.root_dir / "page1.html", "r") as alg_file: - soup = BeautifulSoup(alg_file.read(), "html.parser") + soup = BeautifulSoup(alg_file.read(), "html5lib") num_pages_elem = soup.select("span[data-total-pages]")[0].attrs num_pages = int(num_pages_elem["data-total-pages"]) @@ -89,7 +89,7 @@ class FIPSAlgorithmDataset(Dataset, ComplexSerializableType): continue with open(f, "r", encoding="utf-8") as handle: - html_soup = BeautifulSoup(handle.read(), "html.parser") + html_soup = BeautifulSoup(handle.read(), "html5lib") table = html_soup.find("table", class_="table table-condensed publications-table table-bordered") tbody_contents = table.find("tbody").find_all("tr") diff --git a/sec_certs/sample/fips.py b/sec_certs/sample/fips.py index ed9a90f8..723bc332 100644 --- a/sec_certs/sample/fips.py +++ b/sec_certs/sample/fips.py @@ -426,7 +426,7 @@ class FIPSCertificate(Certificate["FIPSCertificate", "FIPSCertificate.Heuristics items_found["cert_id"] = int(file.stem) text = sec_certs.utils.extract.load_cert_html_file(str(file)) - soup = BeautifulSoup(text, "html.parser") + soup = BeautifulSoup(text, "html5lib") for div in soup.find_all("div", class_="row padrow"): _FIPSHTMLParser.parse_html_main(div, items_found) diff --git a/sec_certs/sample/fips_iut.py b/sec_certs/sample/fips_iut.py index 2fab9a56..6cdc5900 100644 --- a/sec_certs/sample/fips_iut.py +++ b/sec_certs/sample/fips_iut.py @@ -71,7 +71,7 @@ class IUTSnapshot(ComplexSerializableType): def from_page(cls, content: bytes, snapshot_date: datetime) -> "IUTSnapshot": if not content: raise ValueError("Empty content in IUT.") - soup = BeautifulSoup(content, "html.parser") + soup = BeautifulSoup(content, "html5lib") tables = soup.find_all("table") if len(tables) != 1: raise ValueError("Not only a single table in IUT.") diff --git a/sec_certs/sample/fips_mip.py b/sec_certs/sample/fips_mip.py index 4bb33904..71c2d9d2 100644 --- a/sec_certs/sample/fips_mip.py +++ b/sec_certs/sample/fips_mip.py @@ -159,7 +159,7 @@ class MIPSnapshot(ComplexSerializableType): def from_page(cls, content: bytes, snapshot_date: datetime) -> "MIPSnapshot": if not content: raise ValueError("Empty content in MIP.") - soup = BeautifulSoup(content, "html.parser") + soup = BeautifulSoup(content, "html5lib") tables = soup.find_all("table") if len(tables) != 1: raise ValueError("Not only a single table in MIP data.") |
