aboutsummaryrefslogtreecommitdiffhomepage
path: root/sec_certs
diff options
context:
space:
mode:
authorStanislav Boboň2021-05-11 17:47:48 +0200
committerStanislav Boboň2021-05-11 17:47:48 +0200
commitffafebbabcd0aac1deba8dca7cd4c13b0cbbb031 (patch)
tree7fcbc40b34b86ee13daec326dde99811fc5e8d21 /sec_certs
parent6c68b465bc592b517c4ebdfacbd977502647cbdb (diff)
parente3c002a63725e9e79ce81a09a7c7055c61ba5010 (diff)
downloadsec-certs-ffafebbabcd0aac1deba8dca7cd4c13b0cbbb031.tar.gz
sec-certs-ffafebbabcd0aac1deba8dca7cd4c13b0cbbb031.tar.zst
sec-certs-ffafebbabcd0aac1deba8dca7cd4c13b0cbbb031.zip
Merge branch 'dev' into fips
Diffstat (limited to 'sec_certs')
-rw-r--r--sec_certs/certificate.py7
-rw-r--r--sec_certs/constants.py2
-rw-r--r--sec_certs/dataset.py12
-rwxr-xr-xsec_certs/entrypoints/fips_certificates.py212
-rw-r--r--sec_certs/entrypoints/process_certificates.py333
5 files changed, 557 insertions, 9 deletions
diff --git a/sec_certs/certificate.py b/sec_certs/certificate.py
index 12cb6adc..85ab5344 100644
--- a/sec_certs/certificate.py
+++ b/sec_certs/certificate.py
@@ -905,20 +905,23 @@ class CommonCriteriaCert(Certificate, ComplexSerializableType):
cpe_matches: Optional[List[Tuple[float, CPE]]]
verified_cpe_matches: Optional[List[CPE]]
related_cves: Optional[List[str]]
+ labeled: bool
def __init__(self,
extracted_versions: Optional[List[str]] = None,
cpe_matches: Optional[List[str]] = None,
verified_cpe_matches: Optional[List[str]] = None,
- related_cves: Optional[List[CVE]] = None):
+ related_cves: Optional[List[CVE]] = None,
+ labeled: bool = False):
self.extracted_versions = extracted_versions
self.cpe_matches = cpe_matches
self.cpe_candidate_vendors = None
self.verified_cpe_matches = verified_cpe_matches
self.related_cves = related_cves
+ self.labeled = labeled
def to_dict(self):
- return {'extracted_versions': self.extracted_versions, 'cpe_matches': self.cpe_matches, 'verified_cpe_matches': self.verified_cpe_matches, 'related_cves': self.related_cves}
+ return {'extracted_versions': self.extracted_versions, 'cpe_matches': self.cpe_matches, 'verified_cpe_matches': self.verified_cpe_matches, 'related_cves': self.related_cves, 'labeled': self.labeled}
@classmethod
def from_dict(cls, dct: Dict[str, str]):
diff --git a/sec_certs/constants.py b/sec_certs/constants.py
index 4a4cf577..99d81640 100644
--- a/sec_certs/constants.py
+++ b/sec_certs/constants.py
@@ -7,7 +7,7 @@ RETURNCODE_NOK = 'nok'
REQUEST_TIMEOUT = 10
CPE_MATCHING_THRESHOLD = 70
-CPE_MAX_MATCHES = 10
+CPE_MAX_MATCHES = 20
MIN_CORRECT_CERT_SIZE = 5000
diff --git a/sec_certs/dataset.py b/sec_certs/dataset.py
index b80b25da..7369cdf1 100644
--- a/sec_certs/dataset.py
+++ b/sec_certs/dataset.py
@@ -735,7 +735,7 @@ class CCDataset(Dataset, ComplexSerializableType):
for i, x in enumerate(certificates_to_verify):
print(f'\n[{i}/{n_certs_to_verify}] Vendor: {x.manufacturer}, Name: {x.name}')
for index, c in enumerate(x.heuristics.cpe_matches):
- print(f'\t- {[index]}: {c[1]}')
+ print(f'\t- {[index]}: {c[1].vendor} {c[1].title} CPE-URI: {c[1].uri}')
print(f'\t- [A]: All are fitting')
print(f'\t- [X]: No fitting match')
inpts = input('Select fitting CPE matches (split with comma if choosing more):').strip().split(',')
@@ -757,12 +757,12 @@ class CCDataset(Dataset, ComplexSerializableType):
matches = [x.heuristics.cpe_matches[y][1] for y in inpts]
self[x.dgst].heuristics.verified_cpe_matches = matches
- if i != 0 and not i % 10 and update_json:
- print(f'Saving progress.')
- self.to_json()
+ if i != 0 and not i % 10 and update_json:
+ print(f'Saving progress.')
+ self.to_json()
+ self[x.dgst].heuristics.labeled = True
- certs_to_verify: List[CommonCriteriaCert] = [x for x in self if (
- x.heuristics.cpe_matches and not x.heuristics.verified_cpe_matches)]
+ certs_to_verify: List[CommonCriteriaCert] = [x for x in self if (x.heuristics.cpe_matches and not x.heuristics.labeled)]
logger.info('Manually verifying CPE matches')
time.sleep(0.05) # easier than flushing the logger
verify_certs(certs_to_verify)
diff --git a/sec_certs/entrypoints/fips_certificates.py b/sec_certs/entrypoints/fips_certificates.py
new file mode 100755
index 00000000..eb147d48
--- /dev/null
+++ b/sec_certs/entrypoints/fips_certificates.py
@@ -0,0 +1,212 @@
+#!/usr/bin/env python3
+import json
+import os
+import re
+import time
+from pathlib import Path
+from typing import Set, Optional, List, Dict
+from bs4 import BeautifulSoup
+
+from graphviz import Digraph
+import click
+import pikepdf
+from tabula import read_pdf
+
+from sec_certs.download import download_fips_web, download_fips
+from sec_certs import extract_certificates
+from sec_certs.files import load_json_files, FILE_ERRORS_STRATEGY, search_files
+
+FIPS_BASE_URL = 'https://csrc.nist.gov'
+FIPS_MODULE_URL = 'https://csrc.nist.gov/projects/cryptographic-module-validation-program/certificate/'
+
+
+def extract_filename(file: str) -> str:
+ """
+ Extracts filename from path
+ @param file: UN*X path
+ :return: filename without last extension
+ """
+ return os.path.splitext(os.path.basename(file))[0]
+
+
+def initialize_entry(current_items_found):
+ pass
+
+
+def remove_algorithms_from_extracted_data(items, html):
+ pass
+
+
+def validate_results(items: Dict, html: Dict):
+ """
+ Function that validates results and finds the final connection output
+ :param items: All keyword items found in pdf files
+ :param html: All items extracted from html files - this is where we store connections
+ """
+ broken_files = set()
+ for file_name in items:
+ for rule in items[file_name]['rules_cert_id']:
+ for cert in items[file_name]['rules_cert_id'][rule]:
+ cert_id = ''.join(filter(str.isdigit, cert))
+
+ if cert_id == '' or cert_id not in html:
+ # TEST
+ # if cert_id == '' or int(cert_id) > 3730:
+ broken_files.add(file_name)
+ items[file_name]['file_status'] = False
+ html[file_name]['file_status'] = False
+ break
+ if broken_files:
+ print("WARNING: CERTIFICATE FILES WITH WRONG CERTIFICATES PARSED")
+ print(*sorted(list(broken_files)), sep='\n')
+ print("... skipping these...")
+ print("Total non-analyzable files:", len(broken_files))
+
+ for file_name in items:
+ html[file_name]['Connections'] = []
+ if not items[file_name]['file_status']:
+ continue
+ if items[file_name]['rules_cert_id'] == {}:
+ continue
+ for rule in items[file_name]['rules_cert_id']:
+ for cert in items[file_name]['rules_cert_id'][rule]:
+ cert_id = ''.join(filter(str.isdigit, cert))
+ if cert_id not in html[file_name]['Connections']:
+ html[file_name]['Connections'].append(cert_id)
+
+
+def parse_list_of_tables(txt: str) -> Set[str]:
+ pass
+
+
+def extract_page_number(txt: str) -> Optional[str]:
+ """
+ Parses chunks of text that are supposed to be mentioning table and having a footer
+ :param txt: input chunk
+ :return: page number
+ """
+ # Page # of #
+ m = re.findall(r"(?P<pattern>(?:[Pp]age) (?P<page_num>\d+)(?: of \d+))", txt)
+ if m:
+ return m[-1][-1]
+ # Page #
+ m = re.findall(r"(?P<pattern>(?:[Pp]age) (?P<page_num>\d+)(?: of \d+)?)", txt)
+ if m:
+ return m[-1][-1]
+ # # of #
+ m = re.findall(r"(?P<pattern>(?:[Pp]age)? ?(?P<page_num>\d+)(?: of \d+))", txt)
+ if m:
+ return m[-1][-1]
+ # number alone
+ m = re.findall(r"(?P<pattern>(?:[Pp]age)? ?(?P<page_num>\d+)(?: of \d+)?)", txt)
+ return m[-1][-1] if m else None
+
+
+def find_tables_iterative(file_text: str) -> List[int]:
+ pass
+
+
+def find_footers(txt: str, num_pages: int) -> Optional[List]:
+ footer_regex = re.compile(
+ r"(?:Table[^\f]*)(?P<first>^[\S\t ]*$)\n(?P<second>(\f[ \t\S]+)$)(?P<third>\n^[ \t\S]+?$)?",
+ re.MULTILINE)
+
+ # We have 2 groups, one is optional - trying to parse 2 lines (just in case)
+ footer1 = [m.group('first') for m in footer_regex.finditer(txt)]
+ footer2 = [m.group('second') for m in footer_regex.finditer(txt)]
+ footer3 = [m.group('third') for m in footer_regex.finditer(txt)]
+
+ # if len(footer2) < len(footer1):
+ # footer2 += [''] * (len(footer1) - len(footer2))
+
+ # zipping them together
+ footer_complete = [m[0] + m[1] + m[2] for m in zip(footer1, footer2, footer3) if
+ m[0] is not None and m[1] is not None and m[2] is not None]
+
+ # removing None and duplicates
+ footers = [extract_page_number(x) for x in footer_complete]
+ footers = list(dict.fromkeys([x for x in footers if x is not None and 0 < int(x) < num_pages]))
+ print(footers)
+ if footers:
+ return footers
+
+
+def find_tables(txt: str, file_name: Path) -> Optional[List]:
+ pass
+
+
+def parse_algorithms(a, b=False):
+ pass
+
+
+def extract_certs_from_tables(list_of_files: List, html_items: Dict) -> List[Path]:
+ pass
+
+
+@click.command()
+@click.argument("directory", required=True, type=str)
+@click.option("--do-download-meta", "do_download_meta", is_flag=True)
+@click.option("--do-download-certs", "do_download_certs", is_flag=True)
+@click.option("-t", "--threads", "threads", type=int, default=4)
+def main(directory, do_download_meta: bool, do_download_certs: bool, threads: int):
+ start = time.time()
+ directory = Path(directory)
+ web_dir = directory / "web"
+ fragments_dir = directory / "fragments"
+ results_dir = directory / "results"
+ policies_dir = directory / "security_policies"
+
+ directory.mkdir(parents=True, exist_ok=True)
+ web_dir.mkdir(parents=True, exist_ok=True)
+ fragments_dir.mkdir(parents=True, exist_ok=True)
+ results_dir.mkdir(parents=True, exist_ok=True)
+ policies_dir.mkdir(parents=True, exist_ok=True)
+
+ if do_download_meta:
+ download_fips_web(web_dir)
+
+ if do_download_certs:
+ download_fips(web_dir, policies_dir, threads)
+
+ print(f"Missing security policies: Total {len([])}")
+ print(f"Not available security policies: Total {len([])}")
+ files_to_load = [
+ results_dir / 'fips_data_keywords_all.json',
+ results_dir / 'fips_html_all.json'
+ ]
+
+ for file in files_to_load:
+ if not os.path.isfile(file):
+ items = extract_certificates.extract_certificates_keywords(
+ policies_dir,
+ fragments_dir, 'fips', fips_items=None,
+ should_censure_right_away=True)
+ with open(results_dir / 'fips_data_keywords_all.json', 'w') as f:
+ json.dump(items, f, indent=4, sort_keys=True)
+ break
+
+ print("EXTRACTION DONE")
+ items, html = load_json_files(files_to_load)
+
+ print("FINDING TABLES")
+ not_decoded = extract_certs_from_tables(search_files(policies_dir), html)
+
+ print("NOT DECODED:", not_decoded)
+ with open(results_dir / 'broken_files.json', 'w') as f:
+ json.dump(not_decoded, f)
+
+ print("REMOVING ALGORITHMS")
+ remove_algorithms_from_extracted_data(items, html)
+
+ print("VALIDATING RESULTS")
+ validate_results(items, html)
+ with open(results_dir / 'fips_html_all.json', 'w') as f:
+ json.dump(html, f, indent=4, sort_keys=True)
+ print("PLOTTING GRAPH")
+ # get_dot_graph(html, results_dir / 'output')
+ end = time.time()
+ print("TIME:", end - start)
+
+
+if __name__ == '__main__':
+ main()
diff --git a/sec_certs/entrypoints/process_certificates.py b/sec_certs/entrypoints/process_certificates.py
new file mode 100644
index 00000000..5272fc67
--- /dev/null
+++ b/sec_certs/entrypoints/process_certificates.py
@@ -0,0 +1,333 @@
+#!/usr/bin/env python3
+
+import click
+from sec_certs.files import load_json_files
+from sec_certs.extract_certificates import *
+from sec_certs.analyze_certificates import *
+from sec_certs.download import download_cc_web, download_cc
+from sec_certs.cert_rules import rules as cc_search_rules
+
+
+@click.command()
+@click.argument("directory", required=True, type=str)
+@click.option("--fresh", "do_complete_extraction", is_flag=True, help="Whether to extract from a fresh state.")
+@click.option("--do-download-meta", "do_download_meta", is_flag=True, help="Whether to download meta pages.")
+@click.option("--do-extraction-meta", "do_extraction_meta", is_flag=True, help="Whether to extract information from the meta pages.")
+@click.option("--do-download-certs", "do_download_certs", is_flag=True, help="Whether to download certs.")
+@click.option("--do-pdftotext", "do_pdftotext", is_flag=True, help="Whether to perform pdftotext conversion of the certs.")
+@click.option("--do-extraction", "do_extraction_certs", is_flag=True, help="Whether to extract information from the certs.")
+@click.option("--do-pairing", "do_pairing", is_flag=True, help="Whether to pair PP stuff.")
+@click.option("--do-processing", "do_processing", is_flag=True, help="Whether to process certificates.")
+@click.option("--do-analysis", "do_analysis", is_flag=True, help="Whether to analyse certificates.")
+@click.option("--do-analysis-fips", "do_analysis_fips", is_flag=True, help="Whether to analyse fips certificates.")
+@click.option("--do-find-affected", "do_find_affected", help="Find affected certs.", multiple=True, type=str, metavar="certificate id")
+@click.option("--do-find-affecting", "do_find_affecting", help="Find certificates affecting the provided one", multiple=True, type=str, metavar="certificate id")
+@click.option("--do-find-affected-keyword", "do_find_affected_keywords", help="Find certs referencing all certs with specific keyword.", multiple=True, type=str, metavar="keyword")
+@click.option("--analysis-label", "analysis_label", help="Optional custom label for analysis results", multiple=False, type=str, metavar="cutsom label")
+@click.option("-t", "--threads", "threads", type=int, default=4, help="Amount of threads to use.")
+def main(directory, do_complete_extraction: bool, do_download_meta: bool, do_extraction_meta: bool,
+ do_download_certs: bool, do_pdftotext: bool, do_extraction_certs: bool,
+ do_pairing: bool, do_processing: bool, do_analysis: bool, do_analysis_fips: bool, do_find_affected: list,
+ do_find_affected_keywords: list, do_find_affecting: list, analysis_label: str, threads: int):
+
+ directory = Path(directory)
+ web_dir = directory / "web"
+ walk_dir = directory / "certs"
+ certs_dir = walk_dir / "certs"
+ targets_dir = walk_dir / "targets"
+ pp_dir = directory / "pp"
+ fragments_dir = directory / "cert_fragments"
+ pp_fragments_dir = directory / "pp_fragments"
+ results_dir = directory / "results"
+
+ web_dir.mkdir(parents=True, exist_ok=True)
+ walk_dir.mkdir(parents=True, exist_ok=True)
+ certs_dir.mkdir(parents=True, exist_ok=True)
+ targets_dir.mkdir(parents=True, exist_ok=True)
+ pp_dir.mkdir(parents=True, exist_ok=True)
+ fragments_dir.mkdir(parents=True, exist_ok=True)
+ pp_fragments_dir.mkdir(parents=True, exist_ok=True)
+ results_dir.mkdir(parents=True, exist_ok=True)
+
+ #
+ # Start processing
+ #
+ do_analysis_filtered = True
+
+ if do_complete_extraction:
+ # analyze all files from scratch, set 'previous' state to empty dict
+ prev_csv = {}
+ prev_html = {}
+ prev_download = []
+ prev_front = {}
+ prev_keywords = {}
+ prev_pdf_meta = {}
+ else:
+ # load previously analyzed results
+ prev_csv, prev_html, prev_download, prev_front, prev_keywords, prev_pdf_meta = load_json_files(
+ map(lambda x: results_dir / x, ['certificate_data_csv_all.json',
+ 'certificate_data_html_all.json',
+ 'certificate_data_download_all.json',
+ 'certificate_data_frontpage_all.json',
+ 'certificate_data_keywords_all.json',
+ 'certificate_data_pdfmeta_all.json']))
+
+ if do_download_meta:
+ download_cc_web(web_dir, threads)
+
+ # NOTE: Code below is preparation for differetian download of only new certificates
+ # - unfinished now
+ # print('*** Items: {} vs. {}'.format(len(current_html.keys()), len(prev_html.keys())))
+ # current_html_keys = sorted(current_html.keys())
+ # prev_html_keys = sorted(prev_html.keys())
+ # new_items = list(set(current_html.keys()) - set(prev_html.keys()))
+ # print('*** New items detected: {}'.format(len(new_items)))
+ #
+ # # find new items which are not yet processed based on the value of raw csv line
+ # new_items = []
+ # for current_item_key in current_csv.keys():
+ # current_item = current_csv[current_item_key]
+ # current_raw_csv = current_item['csv_scan']['raw_csv_line']
+ # match_found = False
+ # for prev_item_key in prev_csv.keys():
+ # prev_item = prev_csv[prev_item_key]
+ # prev_raw_csv = prev_item['csv_scan']['raw_csv_line']
+ # if current_raw_csv == prev_raw_csv:
+ # match_found = True
+ # break
+ # if not match_found:
+ # # we found new item
+ # new_items.append(current_item_key)
+ #
+ # print('*** New items detected: {}'.format(len(new_items)))
+
+ if do_extraction_meta:
+ all_csv = extract_certificates_csv(web_dir)
+ all_html, certs, updates = extract_certificates_html(web_dir)
+
+ with open(results_dir / "certificate_data_csv_all.json", "w") as write_file:
+ json.dump(all_csv, write_file, indent=4, sort_keys=True)
+ with open(results_dir / "certificate_data_html_all.json", "w") as write_file:
+ json.dump(all_html, write_file, indent=4, sort_keys=True)
+ with open(results_dir / "certificate_data_download_all.json", "w") as write_file:
+ json.dump(certs + updates, write_file, indent=4, sort_keys=True)
+
+ if do_download_certs:
+ all_download = load_json_files([results_dir / "certificate_data_download_all.json"])
+ download_cc(walk_dir, all_download[0], threads)
+
+ if do_pdftotext:
+ convert_pdf_files(walk_dir, threads, ["-raw"])
+
+ if do_extraction_certs:
+ all_keywords = extract_certificates_keywords_parallel(walk_dir, fragments_dir, 'certificate', cc_search_rules, threads)
+ with open(results_dir / "certificate_data_keywords_all.json", "w") as write_file:
+ json.dump(all_keywords, write_file, indent=4, sort_keys=True)
+
+ all_front = extract_certificates_frontpage(walk_dir)
+ with open(results_dir / "certificate_data_frontpage_all.json", "w") as write_file:
+ json.dump(all_front, write_file, indent=4, sort_keys=True)
+
+ all_pdf_meta = extract_certificates_pdfmeta_parallel(walk_dir, 'certificate', threads)
+ with open(results_dir / "certificate_data_pdfmeta_all.json", "w") as write_file:
+ json.dump(all_pdf_meta, write_file, indent=4, sort_keys=True)
+
+
+ # if do_extraction_pp:
+ # all_pp_csv = extract_protectionprofiles_csv(web_dir)
+ # all_pp_front = extract_protectionprofiles_frontpage(pp_dir)
+ # all_pp_keywords = extract_certificates_keywords(pp_dir, pp_fragments_dir, 'pp')
+ # all_pp_pdf_meta = extract_certificates_pdfmeta(pp_dir, 'pp')
+ #
+ # # save joined results
+ # with open("pp_data_csv_all.json", "w") as write_file:
+ # write_file.write(json.dumps(all_pp_csv, indent=4, sort_keys=True))
+ # with open("pp_data_frontpage_all.json", "w") as write_file:
+ # write_file.write(json.dumps(all_pp_front, indent=4, sort_keys=True))
+ # with open("pp_data_keywords_all.json", "w") as write_file:
+ # write_file.write(json.dumps(all_pp_keywords, indent=4, sort_keys=True))
+ # with open("pp_data_pdfmeta_all.json", "w") as write_file:
+ # write_file.write(json.dumps(all_pp_pdf_meta, indent=4, sort_keys=True))
+
+ if do_pairing:
+ # # PROTECTION PROFILES
+ # # load results from previous step
+ # all_pp_csv, all_pp_front, all_pp_keywords, all_pp_pdf_meta = load_json_files(
+ # ['pp_data_csv_all.json', 'pp_data_frontpage_all.json',
+ # 'pp_data_keywords_all.json', 'pp_data_pdfmeta_all.json'])
+ # # check for unexpected results
+ # check_expected_pp_results({}, all_pp_csv, {}, all_pp_keywords)
+ # # collate all results into single file
+ # all_pp_items = collate_certificates_data({}, all_pp_csv, all_pp_front, all_pp_keywords, all_pp_pdf_meta, 'link_pp_document')
+ # # write collated result
+ # with open("pp_data_complete.json", "w") as write_file:
+ # write_file.write(json.dumps(all_pp_items, indent=4, sort_keys=True))
+
+ # CERTIFICATES
+ # load results from previous step
+ all_csv, all_html, all_front, all_keywords, all_pdf_meta = load_json_files(
+ map(lambda x: results_dir / x, ['certificate_data_csv_all.json',
+ 'certificate_data_html_all.json',
+ 'certificate_data_frontpage_all.json',
+ 'certificate_data_keywords_all.json',
+ 'certificate_data_pdfmeta_all.json']))
+ # check for unexpected results
+ check_expected_cert_results(all_html, all_csv, all_front, all_keywords, all_pdf_meta)
+ # collate all results into single file
+ all_cert_items = collate_certificates_data(all_html, all_csv, all_front, all_keywords, all_pdf_meta, 'link_security_target')
+
+ # write collated result
+ with open(results_dir / "certificate_data_complete.json", "w") as write_file:
+ json.dump(all_cert_items, write_file, indent=4, sort_keys=True)
+
+ if do_processing:
+ # load information about protection profiles as extracted by sec-certs-pp tool
+ all_pp_items = {}
+ with open(results_dir / 'pp_data_complete_processed.json') as json_file:
+ all_pp_items = json.load(json_file)
+
+ with open(results_dir / 'certificate_data_complete.json') as json_file:
+ all_cert_items = json.load(json_file)
+
+ all_cert_items = process_certificates_data(all_cert_items, all_pp_items)
+
+ with open(results_dir / "certificate_data_complete_processed.json", "w") as write_file:
+ json.dump(all_cert_items, write_file, indent=4, sort_keys=True)
+
+ if do_analysis:
+ with open(results_dir / 'certificate_data_complete_processed.json') as json_file:
+ all_cert_items = json.load(json_file)
+
+ if do_analysis_filtered:
+ # plot only selected analysis up to date 2020
+ do_analysis_force_end_date(all_cert_items, results_dir, 2020)
+
+ # analyze only smartcards
+ do_analysis_only_filtered(all_cert_items, results_dir,
+ ['csv_scan', 'cc_category'], 'ICs, Smart Cards and Smart Card-Related Devices and Systems')
+ # analyze only operating systems
+ do_analysis_only_filtered(all_cert_items, results_dir,
+ ['csv_scan', 'cc_category'], 'Operating Systems')
+
+ # analyze separate manufacturers
+ do_analysis_manufacturers(all_cert_items, results_dir)
+
+ # archived on 09/01/2019
+ do_analysis_09_01_2019_archival(all_cert_items, results_dir)
+
+ # analyze all certificates together
+ do_analysis_everything(all_cert_items, results_dir)
+
+ with open(results_dir / "certificate_data_complete_processed_analyzed.json", "w") as write_file:
+ json.dump(all_cert_items, write_file, indent=4, sort_keys=True)
+
+ # example: --do-find-affected-keyword v1\.02\.013 --analysis-label roca # (roca library)
+ # example: --do-find-affected-keyword AT90SC --do-find-affected-keyword 00\.03\.11\.05 --analysis-label minerva # (minerva library and chip)
+ # note: keyword search is as by regexes, so mind . etc.
+ if len(do_find_affected_keywords) > 0:
+ results_dir = results_dir \
+
+ search_rules = {'keyword': do_find_affected_keywords}
+ all_keywords = extract_certificates_keywords_parallel(walk_dir, None, 'certificate', search_rules, threads)
+
+ # extract file names with keyword(s) match, extract cert id(s), fill do_find_affected list for further analysis
+ with open(results_dir / 'certificate_data_complete_processed.json') as json_file:
+ all_cert_items = json.load(json_file)
+
+ # match search results to cert ids
+ certs_with_keywords = process_matched_keywords(all_cert_items, all_keywords,
+ list(do_find_affected_keywords), results_dir)
+
+ # save list of found cert ids to separate json
+ name_results = get_name_for_keyword_search_results(do_find_affected_keywords)
+ file_name_results = analysis_label + '_' + name_results + '.json'
+ with open(results_dir / file_name_results, "w") as write_file:
+ json.dump(certs_with_keywords, write_file, indent=4, sort_keys=True)
+
+ # populate list with certs is to analyse (same as would be --do-find-affected with explicitly specified ids)
+ for i in certs_with_keywords['certs'].keys():
+ do_find_affected = do_find_affected + (i,)
+
+ # set output folder according to analysis label
+ out_folder = analysis_label + '_' + name_results
+ results_out_dir = results_dir / out_folder
+
+ do_analysis_affected(all_cert_items, results_out_dir, list(do_find_affected), analysis_label)
+
+ # analysis of all certs referencing (directly/indirectly) the specified cert id(s)
+ # example: --do-find-affected BSI-DSZ-CC-0782-2012
+ # example: --do-find-affected BSI-DSZ-CC-0833-2013 --do-find-affected BSI-DSZ-CC-0921-2014 --analysis-label roca_ATeHealth_Atos # (from eIDAS ID163484)
+ # example: --do-find-affected BSI-DSZ-CC-0758-2012 --do-find-affected BSI-DSZ-CC-0782-2012 --analysis-label roca_ATeHealth_Inf
+ if len(do_find_affected) > 0:
+ with open(results_dir / 'certificate_data_complete_processed_analyzed.json') as json_file:
+ all_cert_items = json.load(json_file)
+ # set output folder according to analysis label
+ results_out_dir = results_dir / analysis_label
+ do_analysis_affected(all_cert_items, results_out_dir, list(do_find_affected), analysis_label)
+
+ # find all certificates which are potentially affecting security of the provided one ()
+ # example: --do-find-affecting ANSSI-CC-2013/55 # Estonia estID
+ # example: --do-find-affecting ANSSI-CC-2020/44 # eTravel v2.2 EAC/BAC on MultiApp v4.0.1 platform with Filter Set 1.0 version 1.0
+ if len(do_find_affecting) > 0:
+ with open(results_dir / 'certificate_data_complete_processed_analyzed.json') as json_file:
+ all_cert_items = json.load(json_file)
+ # set output folder according to analysis label
+ results_out_dir = results_dir / analysis_label
+ do_analysis_affecting(all_cert_items, results_out_dir, list(do_find_affecting), analysis_label)
+
+ # analysis of fips extracted data
+ if do_analysis_fips:
+ with open(results_dir / 'fips_full_dataset.json') as json_file:
+ all_cert_items = json.load(json_file)
+
+ # idea: transform into cc-like json then use same analysis functions
+ do_analysis_fips_certs(all_cert_items, results_dir)
+
+
+if __name__ == "__main__":
+ main()
+
+
+ # TODO
+ # add saving of logs into file
+ # include parsing from protection profiles repo
+ # add differential partial download of new files only + processing + combine
+ # generate download script only for new files (need to have previous version of files stored)
+ # option for extraction of info just for single file?
+ # allow for late extraction of keywords (only newly added regexes)
+ # extraction of keywords done with the provided cert_rules_dict => cert_rules.py and cert_rules_new.py
+ # detect archival of certificates
+ # add tests - few selected files
+ # add detection of overly long regex matches
+ # add analysis of target CC version
+ # extract even more pdf file metadata https://github.com/pdfminer/pdfminer.six
+ # protection profiles dependency graph similarly as certid dependency graph is done
+ # If None == protection profile => Match PP with its assurance level and recompute
+ # extract info about protection profiles, download and parse pdf, map to referencing files
+ # analysis of PP only: which PP is the most popular?, what schemes/countries are doing most...
+ # analysis of certificates in time (per year) (different schemes)
+ # how many certificates are extended? How many times
+ # analysis of use of protection profiles
+ # analysis of security targets documents
+ # analysis of big cert clusters
+ # improve logging (info, warnings, errors, final summary)
+ # save as json, named segments (start_segment('name'), end_segment('name'), print('log_line', level)
+ # other schemes: FIPS140-2 certs, EMVCo, Visa Certification, American Express Certification, MasterCard Certification
+ # download and analyse CC documentation
+ # solve treatment of unicode characters
+ # analyze bibliography
+ # Statistics about number of characters (length), words, pages - histogram of pdf length & extracted text length
+ # add keywords extraction for trademarks (e.g, from 0963V2b_pdf.pdf)
+ # FRONTPAGE
+ # extract frontpage also from other than anssi and bsi certificates (US, BE...)
+ # add extraction of frontpage for protection profiles
+ # PORTABILITY
+ # check functionality on Linux (script %%20 expansions..., \\ vs. /)
+ # Add processing of docx files
+ # search for ATR, Response APDU, and custom commands specifying the IC type (CPLC + others)
+ # e.g., KECS-CR-15-105 XSmart e-Passport V1.4 EAC with SAC on M7892(eng).txt
+ # use pdf2text -raw switch to preserve better tables (needs to be checked wrt existing regexes)
+ # add tool for language detection, and if required, use automatic translation into english (https://pypi.org/project/googletrans/)
+ # analyze technical decisions: https://www.niap-ccevs.org/Documents_and_Guidance/view_tds.cfm
+ # extract names of IC (or other devices) from certificates => results in the list of certified chips in smartcards etc.
+ # analyze SARs and SFRs in correlation with specific company (what level of SAR/SFR can company achieve?)