aboutsummaryrefslogtreecommitdiffhomepage
diff options
context:
space:
mode:
authorJ08nY2022-07-16 16:14:38 +0200
committerJ08nY2022-07-16 16:14:38 +0200
commite0a2a1420cf3c21f0522b4de6db28446b19bbb69 (patch)
tree219604f10de4ca97c818290d01cb7ce870778819
parentef2a3defe1138dc15ba959df5a631ed16894df00 (diff)
downloadsec-certs-e0a2a1420cf3c21f0522b4de6db28446b19bbb69.tar.gz
sec-certs-e0a2a1420cf3c21f0522b4de6db28446b19bbb69.tar.zst
sec-certs-e0a2a1420cf3c21f0522b4de6db28446b19bbb69.zip
Use html5lib instead of html.parser.
-rw-r--r--sec_certs/dataset/common_criteria.py6
-rw-r--r--sec_certs/dataset/fips.py2
-rw-r--r--sec_certs/dataset/fips_algorithm.py4
-rw-r--r--sec_certs/sample/fips.py2
-rw-r--r--sec_certs/sample/fips_iut.py2
-rw-r--r--sec_certs/sample/fips_mip.py2
6 files changed, 9 insertions, 9 deletions
diff --git a/sec_certs/dataset/common_criteria.py b/sec_certs/dataset/common_criteria.py
index 378bca61..d53f1e35 100644
--- a/sec_certs/dataset/common_criteria.py
+++ b/sec_certs/dataset/common_criteria.py
@@ -603,7 +603,7 @@ class CCDataset(Dataset[CommonCriteriaCert], ComplexSerializableType):
cat_dict = {x: y for (x, y) in zip(cc_table_ids, cc_categories)}
with file.open("r") as handle:
- soup = BeautifulSoup(handle, "html.parser")
+ soup = BeautifulSoup(handle, "html5lib")
certs = {}
for key, val in cat_dict.items():
@@ -1006,7 +1006,7 @@ class CCSchemeDataset:
resp = conn.get(url, headers={"User-Agent": "seccerts.org"})
if resp.status_code != requests.codes.ok:
raise ValueError(f"Unable to download: status={resp.status_code}")
- return BeautifulSoup(resp.content, "html.parser")
+ return BeautifulSoup(resp.content, "html5lib")
@staticmethod
def get_australia_in_evaluation():
@@ -1500,7 +1500,7 @@ class CCSchemeDataset:
page = pages.pop()
csrf = soup.find("form", id="fm").find("input", attrs={"name": "csrf"})["value"]
resp = session.post(url, data={"csrf": csrf, "selectPage": page, "product_class": product_class})
- soup = BeautifulSoup(resp.content, "html.parser")
+ soup = BeautifulSoup(resp.content, "html5lib")
tbody = soup.find("table", class_="cpl").find("tbody")
for tr in tbody.find_all("tr"):
tds = tr.find_all("td")
diff --git a/sec_certs/dataset/fips.py b/sec_certs/dataset/fips.py
index 64fd636e..1cb67480 100644
--- a/sec_certs/dataset/fips.py
+++ b/sec_certs/dataset/fips.py
@@ -164,7 +164,7 @@ class FIPSDataset(Dataset[FIPSCertificate], ComplexSerializableType):
def _get_certificates_from_html(self, html_file: Path, update: bool = False) -> Set[str]:
logger.info(f"Getting certificate ids from {html_file}")
with open(html_file, "r", encoding="utf-8") as handle:
- html = BeautifulSoup(handle.read(), "html.parser")
+ html = BeautifulSoup(handle.read(), "html5lib")
table = [x for x in html.find(id="searchResultsTable").tbody.contents if x != "\n"]
entries: Set[str] = set()
diff --git a/sec_certs/dataset/fips_algorithm.py b/sec_certs/dataset/fips_algorithm.py
index caaa5a7b..e9e3680c 100644
--- a/sec_certs/dataset/fips_algorithm.py
+++ b/sec_certs/dataset/fips_algorithm.py
@@ -40,7 +40,7 @@ class FIPSAlgorithmDataset(Dataset, ComplexSerializableType):
logger.error("Couldn't download first page of algo dataset")
with open(self.root_dir / "page1.html", "r") as alg_file:
- soup = BeautifulSoup(alg_file.read(), "html.parser")
+ soup = BeautifulSoup(alg_file.read(), "html5lib")
num_pages_elem = soup.select("span[data-total-pages]")[0].attrs
num_pages = int(num_pages_elem["data-total-pages"])
@@ -89,7 +89,7 @@ class FIPSAlgorithmDataset(Dataset, ComplexSerializableType):
continue
with open(f, "r", encoding="utf-8") as handle:
- html_soup = BeautifulSoup(handle.read(), "html.parser")
+ html_soup = BeautifulSoup(handle.read(), "html5lib")
table = html_soup.find("table", class_="table table-condensed publications-table table-bordered")
tbody_contents = table.find("tbody").find_all("tr")
diff --git a/sec_certs/sample/fips.py b/sec_certs/sample/fips.py
index ed9a90f8..723bc332 100644
--- a/sec_certs/sample/fips.py
+++ b/sec_certs/sample/fips.py
@@ -426,7 +426,7 @@ class FIPSCertificate(Certificate["FIPSCertificate", "FIPSCertificate.Heuristics
items_found["cert_id"] = int(file.stem)
text = sec_certs.utils.extract.load_cert_html_file(str(file))
- soup = BeautifulSoup(text, "html.parser")
+ soup = BeautifulSoup(text, "html5lib")
for div in soup.find_all("div", class_="row padrow"):
_FIPSHTMLParser.parse_html_main(div, items_found)
diff --git a/sec_certs/sample/fips_iut.py b/sec_certs/sample/fips_iut.py
index 2fab9a56..6cdc5900 100644
--- a/sec_certs/sample/fips_iut.py
+++ b/sec_certs/sample/fips_iut.py
@@ -71,7 +71,7 @@ class IUTSnapshot(ComplexSerializableType):
def from_page(cls, content: bytes, snapshot_date: datetime) -> "IUTSnapshot":
if not content:
raise ValueError("Empty content in IUT.")
- soup = BeautifulSoup(content, "html.parser")
+ soup = BeautifulSoup(content, "html5lib")
tables = soup.find_all("table")
if len(tables) != 1:
raise ValueError("Not only a single table in IUT.")
diff --git a/sec_certs/sample/fips_mip.py b/sec_certs/sample/fips_mip.py
index 4bb33904..71c2d9d2 100644
--- a/sec_certs/sample/fips_mip.py
+++ b/sec_certs/sample/fips_mip.py
@@ -159,7 +159,7 @@ class MIPSnapshot(ComplexSerializableType):
def from_page(cls, content: bytes, snapshot_date: datetime) -> "MIPSnapshot":
if not content:
raise ValueError("Empty content in MIP.")
- soup = BeautifulSoup(content, "html.parser")
+ soup = BeautifulSoup(content, "html5lib")
tables = soup.find_all("table")
if len(tables) != 1:
raise ValueError("Not only a single table in MIP data.")