aboutsummaryrefslogtreecommitdiffhomepage
path: root/src
diff options
context:
space:
mode:
authorPetr Svenda2020-01-27 21:05:40 +0100
committerPetr Svenda2020-01-27 21:05:40 +0100
commit5d7673d398d3c6ce0e6fdfe462b6622af498580f (patch)
treeb87798cb420e69e3f88efff4899b60d6ec352723 /src
parent3df5435ee4ffdc62a63e6a80e5f2f267d6ca2a29 (diff)
downloadsec-certs-5d7673d398d3c6ce0e6fdfe462b6622af498580f.tar.gz
sec-certs-5d7673d398d3c6ce0e6fdfe462b6622af498580f.tar.zst
sec-certs-5d7673d398d3c6ce0e6fdfe462b6622af498580f.zip
fixed storage location of final json
Diffstat (limited to 'src')
-rw-r--r--src/process_certificates.py32
1 files changed, 22 insertions, 10 deletions
diff --git a/src/process_certificates.py b/src/process_certificates.py
index 778cf1b9..c82607a2 100644
--- a/src/process_certificates.py
+++ b/src/process_certificates.py
@@ -10,6 +10,7 @@ def do_all_analysis(all_cert_items, filter_label):
analyze_references_graph(['rules_cert_id'], all_cert_items, filter_label)
analyze_eal_frequency(all_cert_items, filter_label)
analyze_sars_frequency(all_cert_items, filter_label)
+ analyze_pdfmeta(all_cert_items, filter_label)
generate_dot_graphs(all_cert_items, filter_label)
plot_certid_to_item_graph(['keywords_scan', 'rules_protection_profiles'], all_cert_items, filter_label, 'certid_pp_graph.dot', False)
@@ -35,9 +36,16 @@ def do_analysis_only_smartcards(all_cert_items, current_dir):
def main():
- # change current directory to store results into results file
current_dir = os.getcwd()
- os.chdir(current_dir + '\\..\\results\\')
+ results_folder = '\\..\\results\\'
+
+ # ensure existence of results folder
+
+ if not os.path.exists(current_dir + results_folder):
+ os.makedirs(current_dir + results_folder)
+
+ # change current directory to store results into results file
+ os.chdir(current_dir + results_folder)
cc_html_files_dir = 'c:\\Certs\\web\\'
@@ -64,9 +72,13 @@ def main():
do_extraction = False
do_extraction_pp = False
do_pairing = False
- do_processing = True
+ do_processing = False
do_analysis = True
+ with open('certificate_data_complete_processed.json') as json_file:
+ all_cert_items = json.load(json_file)
+ analyze_pdfmeta(all_cert_items, '')
+
#all_pp_front = extract_protectionprofiles_frontpage(pp_dir)
#all_pdf_meta = extract_certificates_pdfmeta(walk_dir, 'certificate')
#all_pp_pdf_meta = extract_certificates_pdfmeta(pp_dir, 'pp')
@@ -141,20 +153,18 @@ def main():
# analyze only smartcards
do_analysis_only_smartcards(all_cert_items, current_dir)
+ os.chdir(current_dir)
with open("certificate_data_complete_processed_analyzed.json", "w") as write_file:
write_file.write(json.dumps(all_cert_items, indent=4, sort_keys=True))
- with open('pp_data_complete.json') as json_file:
- all_pp_items = json.load(json_file)
+ #with open('pp_data_complete.json') as json_file:
+ # all_pp_items = json.load(json_file)
# TODO
- # extract pdf file metadata (pdf header, file size...), PyPDF2 https://www.blog.pythonlibrary.org/2018/04/10/extracting-pdf-metadata-and-text-with-python/, https://github.com/pdfminer/pdfminer.six
- # https://pythonhosted.org/PyPDF2/PdfFileReader.html
+ # extract more pdf file metadata https://github.com/pdfminer/pdfminer.six
# allow for late extraction of keywords (only newly added regexes)
- # add extraction of frontpage for protection profiles
# If None == protection profile => Match PP with its assurance level and recompute
- # analyze_sept2019_cleaning(all_cert_items)
# extract info about protection profiles, download and parse pdf, map to referencing files
# analysis of PP only: which PP is the most popular?, what schemes/countries are doing most...
# analysis of certificates in time (per year) (different schemes)
@@ -167,9 +177,11 @@ def main():
# download and analyse CC documentation
# solve treatment of unicode characters
# analyze bibliography
- # histogram/time of cert labs (maybe first two words heuristics)
# Statistics about number of characters (length), words, pages
+ # add keywords extraction for trademarks (e.g, from 0963V2b_pdf.pdf)
+ # FRONTPAGE
# extract frontpage also from other than anssi and bsi certificates (US, BE...)
+ # add extraction of frontpage for protection profiles
if __name__ == "__main__":
main()