diff options
| author | J08nY | 2023-04-13 15:04:05 +0200 |
|---|---|---|
| committer | J08nY | 2023-04-13 15:04:05 +0200 |
| commit | 8e469840c6323fd5a3b140a918fb032cac9bc105 (patch) | |
| tree | 069b07f50f141aba92164da75d16fa92ef1c3566 /src | |
| parent | f98c5d8096b1b8f01bc1fbc9fb01bdd800088ae0 (diff) | |
| download | sec-certs-8e469840c6323fd5a3b140a918fb032cac9bc105.tar.gz sec-certs-8e469840c6323fd5a3b140a918fb032cac9bc105.tar.zst sec-certs-8e469840c6323fd5a3b140a918fb032cac9bc105.zip | |
Enhance Australian CC collection.
Diffstat (limited to 'src')
| -rw-r--r-- | src/sec_certs/dataset/cc_scheme.py | 49 |
1 files changed, 43 insertions, 6 deletions
diff --git a/src/sec_certs/dataset/cc_scheme.py b/src/sec_certs/dataset/cc_scheme.py index 01781e15..d25ba35d 100644 --- a/src/sec_certs/dataset/cc_scheme.py +++ b/src/sec_certs/dataset/cc_scheme.py @@ -1,5 +1,6 @@ import tempfile from pathlib import Path +from typing import Any from urllib.parse import urljoin import requests @@ -19,8 +20,7 @@ class CCSchemeDataset: return BeautifulSoup(resp.content, "html5lib") @staticmethod - def get_australia_in_evaluation(): - # TODO: Information could be expanded by following url. + def get_australia_in_evaluation(enhanced: bool = True): # noqa: C901 soup = CCSchemeDataset._get_page(constants.CC_AUSTRALIA_INEVAL_URL) header = soup.find("h2", text="Products in evaluation") table = header.find_next_sibling("table") @@ -29,14 +29,51 @@ class CCSchemeDataset: tds = tr.find_all("td") if not tds: continue - cert_url = urljoin(constants.CC_AUSTRALIA_BASE_URL, tds[1].find("a")["href"]) - cert = { + cert: dict[str, Any] = { "vendor": sns(tds[0].text), "product": sns(tds[1].text), - "url": cert_url, + "url": urljoin(constants.CC_AUSTRALIA_BASE_URL, tds[1].find("a")["href"]), "level": sns(tds[2].text), } - print(cert) + if enhanced: + e: dict[str, Any] = {} + cert_page = CCSchemeDataset._get_page(cert["url"]) + article = cert_page.find("article", attrs={"role": "article"}) + blocks = article.find("div").find_all("div", class_="flex", recursive=False) + for h2 in blocks[0].find_all("h2"): + val = sns(h2.find_next_sibling("span").text) + h_text = sns(h2.text) + if not h_text: + continue + if "Version:" in h_text: + e["version"] = val + elif "Product type:" in h_text: + e["product_type"] = val + elif "Product status:" in h_text: + e["product_status"] = val + elif "Assurance level:" in h_text: + e["assurance_level"] = val + sides = blocks[1].find_all("div", recursive=False) + for div in sides[0].find_all("div", recursive=False): + h2 = div.find("h2") + h_text = sns(h2.text) + if not h_text: + continue + if "epl-vendor-token" in div.get("class"): + vendor_address = [h_text] + vendor_address.extend([sns(elem.text) for elem in h2.find_next_siblings("div")]) # type: ignore + e["vendor"] = "\n".join(vendor_address) + else: + val = sns(h2.find_next_sibling("span").text) + if "Evaluation Facility:" in h_text: + e["evaluation_facility"] = val + elif "Certification Progress:" in h_text: + e["certification_progress"] = val + elif "Estimated Approval" in h_text: + e["estimated_approval"] = val + e["contacts"] = [sns(p.text) for p in sides[1].find_all("p")] + e["description"] = sns(blocks[2].find("span").text) + cert["enhanced"] = e results.append(cert) return results |
