andruhovski

How to Automate PDF Processing in Python: A Complete Workflow

Sep 23rd, 2026
18,660
0
Never
Not a member of Pastebin yet? Sign Up, it unlocks many cool features!
Python 7.61 KB | Source Code | 0 0
  1. import csv
  2. import logging
  3. import os
  4. import pathlib
  5. import sys
  6. import tempfile
  7. import time
  8.  
  9. import aspose.pdf as apdf
  10. from dotenv import dotenv_values
  11.  
  12.  
  13. MAX_PDF_BYTES = 1_000_000_000  # 1 GB; reject larger inputs before loading them.
  14. REPORT_COLUMNS = ("input_file", "output_file", "status", "processing_seconds", "error")
  15.  
  16.  
  17. class SkipPDF(Exception):
  18.     """The input cannot be processed under the batch's safety policy."""
  19.  
  20.  
  21. def optimize(input_file: str, output_file: str) -> None:
  22.     try:
  23.         document = apdf.Document(input_file)
  24.     except RuntimeError as exc:
  25.         # Aspose's Python bridge exposes .NET exceptions as RuntimeError.
  26.         if "InvalidPasswordException" in str(exc):
  27.             raise SkipPDF("Password-protected PDF; no password supplied") from exc
  28.         if "InvalidPdfFileFormatException" in str(exc):
  29.             raise ValueError(f"Invalid or corrupted PDF: {exc}") from exc
  30.         raise
  31.     try:
  32.         if document.is_encrypted:
  33.             raise SkipPDF("Encrypted PDF; password-protected files are skipped")
  34.         document.optimize()
  35.         document.save(output_file)
  36.     finally:
  37.         del document
  38.  
  39.  
  40. def configure_logging(logs_folder: pathlib.Path) -> None:
  41.     logs_folder.mkdir(parents=True, exist_ok=True)
  42.     logging.basicConfig(
  43.         filename=str(logs_folder / "processing.log"),
  44.         encoding="utf-8",
  45.         level=logging.INFO,
  46.         format="%(asctime)s %(levelname)s %(message)s",
  47.     )
  48.  
  49.  
  50. def find_pdf_files(input_folder: pathlib.Path) -> list[pathlib.Path]:
  51.     return sorted(
  52.         path
  53.         for path in input_folder.iterdir()
  54.         if path.is_file() and path.suffix.lower() == ".pdf"
  55.     )
  56.  
  57.  
  58. def process_file(input_file: pathlib.Path, output_file: pathlib.Path) -> tuple[str, float, str]:
  59.     """Optimize one PDF, recording failures without interrupting the batch."""
  60.     started_at = time.perf_counter()
  61.     logging.info("Processing %s", input_file.name)
  62.     error = ""
  63.     try:
  64.         if os.path.lexists(output_file):
  65.             raise SkipPDF("Output already exists; left unchanged")
  66.         if input_file.stat().st_size > MAX_PDF_BYTES:
  67.             raise SkipPDF("Input exceeds the 1 GB size limit")
  68.         with input_file.open("rb") as source:
  69.             if b"%PDF-" not in source.read(1024):
  70.                 raise ValueError("Invalid PDF: missing PDF header")
  71.         # Publish only complete PDFs. Hard-link creation cannot overwrite an
  72.         # output created by another process after the existence check.
  73.         with tempfile.TemporaryDirectory(prefix=".pdf-", dir=output_file.parent) as work:
  74.             temporary_output = pathlib.Path(work) / "optimized.pdf"
  75.             optimize(str(input_file), str(temporary_output))
  76.             os.link(temporary_output, output_file)
  77.     except (SkipPDF, FileExistsError) as exc:
  78.         status = "skipped"
  79.         error = str(exc)
  80.         logging.warning("Skipped %s: %s", input_file.name, error)
  81.     except PermissionError as exc:
  82.         status = "failed"
  83.         error = f"Permission denied reading input or writing output: {exc}"
  84.         logging.error("Failed to process %s: %s", input_file.name, error)
  85.     except Exception as exc:
  86.         status = "failed"
  87.         error = str(exc)
  88.         logging.exception("Failed to process %s", input_file.name)
  89.     else:
  90.         status = "success"
  91.         logging.info("Successfully processed %s", input_file.name)
  92.     return status, time.perf_counter() - started_at, error
  93.  
  94.  
  95. def process_files(
  96.     input_files: list[pathlib.Path],
  97.     output_folder: pathlib.Path,
  98.     report_path: pathlib.Path,
  99. ) -> tuple[int, int, int]:
  100.     """Write a report and return success, failure, and skipped counts."""
  101.     if input_files:
  102.         output_folder.mkdir(parents=True, exist_ok=True)
  103.  
  104.     successfully_processed = 0
  105.     failed = 0
  106.     skipped = 0
  107.     logging.info("Files found: %d", len(input_files))
  108.  
  109.     with report_path.open("w", newline="", encoding="utf-8") as report_file:
  110.         writer = csv.writer(report_file)
  111.         writer.writerow(REPORT_COLUMNS)
  112.         report_file.flush()
  113.  
  114.         for pdf_in_file in input_files:
  115.             pdf_out_file = output_folder / pdf_in_file.name
  116.             status, processing_seconds, error = process_file(pdf_in_file, pdf_out_file)
  117.             if status == "success":
  118.                 successfully_processed += 1
  119.             elif status == "skipped":
  120.                 skipped += 1
  121.             else:
  122.                 failed += 1
  123.  
  124.             writer.writerow(
  125.                 [
  126.                     str(pdf_in_file),
  127.                     str(pdf_out_file),
  128.                     status,
  129.                     f"{processing_seconds:.3f}",
  130.                     error,
  131.                 ]
  132.             )
  133.             report_file.flush()
  134.  
  135.     return successfully_processed, failed, skipped
  136.  
  137.  
  138. def main(
  139.     base_folder: pathlib.Path | None = None,
  140.     license_path: str | None = None,
  141. ) -> int:
  142.     started_at = time.perf_counter()
  143.     config_path = pathlib.Path(__file__).resolve().parent / ".env"
  144.     with config_path.open(encoding="utf-8") as config_file:
  145.         config = dotenv_values(stream=config_file)
  146.     if base_folder is None:
  147.         configured_base_folder = config.get("BASE_FOLDER")
  148.         if not configured_base_folder or not configured_base_folder.strip():
  149.             raise ValueError("BASE_FOLDER must be set in .env")
  150.         base_folder = pathlib.Path(configured_base_folder).expanduser()
  151.         if not base_folder.is_absolute():
  152.             base_folder = config_path.parent / base_folder
  153.     if license_path is None:
  154.         configured_license_path = config.get("LICENSE_PATH")
  155.         if not configured_license_path or not configured_license_path.strip():
  156.             raise ValueError("LICENSE_PATH must be set in .env")
  157.         license_file = pathlib.Path(configured_license_path).expanduser()
  158.         if not license_file.is_absolute():
  159.             license_file = config_path.parent / license_file
  160.         license_path = str(license_file)
  161.     if not base_folder.is_dir():
  162.         raise NotADirectoryError(f"BASE_FOLDER is not a directory: {base_folder}")
  163.     if not pathlib.Path(license_path).is_file():
  164.         raise FileNotFoundError(f"LICENSE_PATH is not a file: {license_path}")
  165.     logs_folder = base_folder / "logs"
  166.     report_path = logs_folder / "processing-report.csv"
  167.     try:
  168.         configure_logging(logs_folder)
  169.         logging.info("Processing started")
  170.         pdf_in_files = find_pdf_files(base_folder / "input")
  171.         if pdf_in_files:
  172.             pdf_license = apdf.License()
  173.             pdf_license.set_license(license_path)
  174.         else:
  175.             logging.info("No PDF files found; nothing to process")
  176.             print("No PDF files found; nothing to process.")
  177.         successfully_processed, failed, skipped = process_files(
  178.             pdf_in_files, base_folder / "output", report_path
  179.         )
  180.     except (OSError, RuntimeError) as exc:
  181.         message = f"Batch could not complete: {exc}"
  182.         logging.error(message)
  183.         print(message, file=sys.stderr)
  184.         raise
  185.  
  186.     total_seconds = time.perf_counter() - started_at
  187.     summary = (
  188.         f"Files found:            {len(pdf_in_files)}\n"
  189.         f"Successfully processed: {successfully_processed}\n"
  190.         f"Failed:                 {failed}\n"
  191.         f"Skipped:                {skipped}\n"
  192.         f"Total processing time: {total_seconds:.1f} seconds\n"
  193.         f"Report: {report_path.relative_to(base_folder).as_posix()}"
  194.     )
  195.     logging.info("Processing finished\n%s", summary)
  196.     print(summary)
  197.     return 1 if failed else 0
  198.  
  199.  
  200. if __name__ == "__main__":
  201.     sys.exit(main())
  202.  
  203.  
Advertisement