2 min read

Bulk PDF Processor

This utility scans a directory for PDF files and processes them (e.g., extracts text) using a progress bar to track status.

Modules Used:

  • argparse: To handle input/output directories.
  • pathlib: To find files recursively.
  • pypdf: To read the PDF content.
  • tqdm: To show a progress bar.

The Code

Save this as pdf_tool.py.

import argparse
from pathlib import Path
from pypdf import PdfReader
from tqdm import tqdm

def process_pdfs(input_dir, keyword):
    # 1. Find all PDF files
    path = Path(input_dir)
    files = list(path.rglob("*.pdf"))

    print(f"Found {len(files)} PDF files in {input_dir}")

    results = []

    # 2. Iterate with Progress Bar
    for pdf_file in tqdm(files, desc="Processing PDFs", unit="file"):
        try:
            reader = PdfReader(pdf_file)

            # Check first page for keyword
            if len(reader.pages) > 0:
                text = reader.pages[0].extract_text()
                if keyword.lower() in text.lower():
                    results.append(pdf_file.name)

        except Exception as e:
            # Don't crash the loop on a bad file, just print error
            tqdm.write(f"Error reading {pdf_file.name}: {e}")

    # 3. Report Results
    print("\n" + "="*30)
    print(f"Files containing '{keyword}':")
    for res in results:
        print(f"- {res}")

if __name__ == "__main__":
    parser = argparse.ArgumentParser(description="Bulk PDF Scanner")
    parser.add_argument("directory", help="Directory to scan")
    parser.add_argument("--search", required=True, help="Keyword to search for in first page")

    args = parser.parse_args()

    process_pdfs(args.directory, args.search)

Usage

python pdf_tool.py ./documents --search "Invoice"

programming/python/python