Python OCR PDF Extraction

Tesseract Installation

python 复制代码
#!/usr/bin/python3
# 
# Python OCR PDF Extraction
# https://github.com/tesseract-ocr/tesseract
#
# sudo apt install tesseract-ocr
# sudo apt install libtesseract-dev
# pip install pytesseract PyPDF2 pdfplumber opencv-python pillow
# pip install pdf2image
# sudo apt-get install poppler-utils
# sudo apt-get install tesseract-ocr-chi-sim  # Simplified Chinese
# sudo apt-get install tesseract-ocr-chi-tra  # Traditional Chinese
# tesseract --list-langs

import pytesseract
from pdf2image import convert_from_path
from PyPDF2 import PdfReader
import cv2
import numpy as np
from PIL import Image

# Path to Tesseract executable (update to match your system)
pytesseract.pytesseract.tesseract_cmd = '/usr/bin/tesseract'

def preprocess_image(pil_image):
    """
    Preprocesses an image for OCR using OpenCV.
    Converts to grayscale, applies thresholding.
    """
    # Convert PIL image to OpenCV format
    open_cv_image = np.array(pil_image)
    # Convert RGB to BGR (OpenCV default format)
    open_cv_image = cv2.cvtColor(open_cv_image, cv2.COLOR_RGB2BGR)
    # Convert to grayscale
    gray_image = cv2.cvtColor(open_cv_image, cv2.COLOR_BGR2GRAY)
    # Apply binary thresholding
    _, thresh_image = cv2.threshold(gray_image, 128, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU)
    return thresh_image

def extract_text_from_pdf(pdf_path):
    # First try extracting text from the PDF directly
    reader = PdfReader(pdf_path)
    text = ""
    for page in reader.pages:
        text += page.extract_text() or ""

    # If no text is extracted, assume it's a scanned PDF and use OCR
    if not text.strip():
        images = convert_from_path(pdf_path)
        for image in images:
            # Preprocess image for better OCR results
            preprocessed_image = preprocess_image(image)
            # Convert OpenCV image back to PIL format for Tesseract
            pil_image = Image.fromarray(preprocessed_image)
            # Perform OCR
            text += pytesseract.image_to_string(pil_image, lang='chi_sim')

    return text

# Example usage
pdf_path = "scan_2025-01-02_09.31.pdf"
extracted_text = extract_text_from_pdf(pdf_path)
print(extracted_text)
相关推荐
梦想的颜色4 小时前
【编程实战】AI 时代 APP 开发全栈硬核指南:技术选型 + AI 架构 + 模型落地全维度决策
python·flutter·react native·react.js·ai·桌面应用·milvus
Madison-No74 小时前
多语言聊天大模型--测试报告
linux·git·python·selenium·jmeter·自动化·postman
鲲鹏ai4 小时前
盈启鲲鹏数字人招商政策
大数据·人工智能·python
荣码4 小时前
从0到1搭一个生产级RAG系统:串联前面21篇所有知识
java·python
程序人生8885 小时前
PDF 转 Word 版式错乱、表格丢失?DocConverter Web 格式互转引擎的保真实践
前端·人工智能·opencv·机器学习·pdf·word
小师兄吃牛肉5 小时前
Java语法 | 多重循环
java·开发语言·python
明如正午6 小时前
用 Python 一键把 ASC 文件转成 BLF(基于 python-can)
python·can·blf·asc
leisoo80976 小时前
融资融券数据怎么查两融指标含义与杠杆观察方法 IG50免费开源股票数据API接口
开发语言·jvm·数据库·python·json
Patrick在香港6 小时前
同一份脚本,mac 正常 Windows 乱码:open() 默认编码实测(附 3.15 终局)
utf-8·windows·python·macos·跨平台·编码·标准库
勿信日志7 小时前
Playwright 元素定位:一个 width>50 过滤器把 42px 的输入框删掉了
python