diff options
| author | GeorgeFI | 2023-09-20 20:53:19 +0200 |
|---|---|---|
| committer | GeorgeFI | 2023-09-20 20:53:19 +0200 |
| commit | 0eaf610cfe6a0a36cfad2fa71b42d1f921b6e49b (patch) | |
| tree | 6b57094a52cfcb3911d5fe0d42ab690f268311b2 /src | |
| parent | 30f74756fa8f87c42f897fa381700784a1e2dec2 (diff) | |
| download | sec-certs-0eaf610cfe6a0a36cfad2fa71b42d1f921b6e49b.tar.gz sec-certs-0eaf610cfe6a0a36cfad2fa71b42d1f921b6e49b.tar.zst sec-certs-0eaf610cfe6a0a36cfad2fa71b42d1f921b6e49b.zip | |
feat: Added pytesseract wrapper
Diffstat (limited to 'src')
| -rw-r--r-- | src/sec_certs/utils/pdf.py | 14 |
1 files changed, 9 insertions, 5 deletions
diff --git a/src/sec_certs/utils/pdf.py b/src/sec_certs/utils/pdf.py index 749a8a5a..b4ec1e94 100644 --- a/src/sec_certs/utils/pdf.py +++ b/src/sec_certs/utils/pdf.py @@ -11,6 +11,7 @@ from typing import Any import pdftotext import pikepdf +import pytesseract from sec_certs import constants from sec_certs.constants import ( @@ -51,13 +52,16 @@ def ocr_pdf_file(pdf_path: Path) -> str: ) if ppm.returncode != 0: raise ValueError(f"pdftoppm failed: {ppm.returncode}") + for ppm_path in map(Path, glob.glob(str(tmppath / "image*.ppm"))): base = ppm_path.with_suffix("") - tes = subprocess.run( - ["tesseract", "-l", "eng+deu+fra", ppm_path, base], stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL - ) - if tes.returncode != 0: - raise ValueError(f"tesseract failed: {tes.returncode}") + content = pytesseract.image_to_string(ppm_path, lang="eng+deu+fra") + + if content: + with Path(base.with_suffix(".txt")).open("w") as file: + file.write(content) + else: + raise ValueError(f"OCR failed for document {ppm_path}. Check document manually") contents = "" |
