Python pytesseract
A Python wrapper for Google's Tesseract-OCR engine, enabling text extraction from images.
Python pytesseract
A Python wrapper for Google's Tesseract-OCR engine, enabling text extraction from images.
Overview
pytesseract provides a simple interface to Tesseract OCR, allowing Python applications to extract text from images, PDFs, and scanned documents. It supports multiple languages, various output formats, and integrates seamlessly with image processing libraries like Pillow and OpenCV.
flowchart LR
A[Input Image] --> B[Preprocessing]
B --> C[pytesseract]
C --> D[Tesseract OCR]
D --> E[Text Output]
subgraph Output Formats
E --> F[Plain Text]
E --> G[Bounding Boxes]
E --> H[Searchable PDF]
E --> I[HOCR/XML]
end
Installation and Setup
Installing Tesseract OCR Engine
# Ubuntu/Debian
sudo apt-get install tesseract-ocr
# macOS
brew install tesseract
# Windows - download installer from GitHub releases
# https://github.com/UB-Mannheim/tesseract/wiki
# Install additional language packs
sudo apt-get install tesseract-ocr-fra tesseract-ocr-deu # French, German
Installing pytesseract
uv add pytesseract pillow opencv-python
Configuration
import pytesseract
from PIL import Image
# Set Tesseract path (required on Windows, optional on Unix)
pytesseract.pytesseract.tesseract_cmd = r'C:\Program Files\Tesseract-OCR\tesseract.exe'
# Verify installation
print(pytesseract.get_tesseract_version())
Image Preprocessing Techniques
Proper preprocessing dramatically improves OCR accuracy. Apply these techniques before extraction.
flowchart TD
A[Original Image] --> B[Grayscale Conversion]
B --> C[Noise Reduction]
C --> D[Thresholding]
D --> E[Deskewing]
E --> F[Scaling]
F --> G[Ready for OCR]
Grayscale Conversion
import cv2
from PIL import Image
import numpy as np
# Using OpenCV
image = cv2.imread('document.png')
grey = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)
# Using Pillow
image = Image.open('document.png').convert('L')
Noise Reduction
# Gaussian blur for general noise
denoised = cv2.GaussianBlur(grey, (5, 5), 0)
# Median blur for salt-and-pepper noise
denoised = cv2.medianBlur(grey, 3)
# Non-local means denoising (best quality, slower)
denoised = cv2.fastNlMeansDenoising(grey, None, 10, 7, 21)
Thresholding
# Simple binary threshold
_, binary = cv2.threshold(grey, 127, 255, cv2.THRESH_BINARY)
# Otsu's automatic thresholding (recommended)
_, binary = cv2.threshold(grey, 0, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU)
# Adaptive thresholding for uneven lighting
adaptive = cv2.adaptiveThreshold(
grey, 255, cv2.ADAPTIVE_THRESH_GAUSSIAN_C, cv2.THRESH_BINARY, 11, 2
)
Deskewing
def deskew(image):
"""Correct image rotation using Hough transform."""
coords = np.column_stack(np.where(image > 0))
angle = cv2.minAreaRect(coords)[-1]
if angle < -45:
angle = -(90 + angle)
else:
angle = -angle
(h, w) = image.shape[:2]
centre = (w // 2, h // 2)
M = cv2.getRotationMatrix2D(centre, angle, 1.0)
rotated = cv2.warpAffine(
image, M, (w, h), flags=cv2.INTER_CUBIC, borderMode=cv2.BORDER_REPLICATE
)
return rotated
Scaling for Small Text
# Scale up images with small text (optimal DPI is 300)
def scale_image(image, scale_factor=2):
width = int(image.shape[1] * scale_factor)
height = int(image.shape[0] * scale_factor)
return cv2.resize(image, (width, height), interpolation=cv2.INTER_CUBIC)
Complete Preprocessing Pipeline
def preprocess_image(image_path):
"""Apply full preprocessing pipeline for optimal OCR."""
# Load image
image = cv2.imread(image_path)
# Convert to grayscale
grey = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)
# Remove noise
denoised = cv2.fastNlMeansDenoising(grey, None, 10, 7, 21)
# Apply adaptive thresholding
binary = cv2.adaptiveThreshold(
denoised, 255, cv2.ADAPTIVE_THRESH_GAUSSIAN_C, cv2.THRESH_BINARY, 11, 2
)
# Deskew if needed
corrected = deskew(binary)
return corrected
OCR Extraction Methods
Basic Text Extraction
import pytesseract
from PIL import Image
# Simple extraction
text = pytesseract.image_to_string(Image.open('document.png'))
print(text)
# From OpenCV image (numpy array)
text = pytesseract.image_to_string(cv2_image)
# From file path directly
text = pytesseract.image_to_string('document.png')
Bounding Box Data
# Get bounding boxes for each character
boxes = pytesseract.image_to_boxes(Image.open('document.png'))
for box in boxes.splitlines():
char, x1, y1, x2, y2, page = box.split()
print(f"Character '{char}' at ({x1}, {y1}) to ({x2}, {y2})")
# Get detailed data including word-level boxes
data = pytesseract.image_to_data(Image.open('document.png'))
print(data)
# As dictionary for easier processing
data_dict = pytesseract.image_to_data(
Image.open('document.png'), output_type=pytesseract.Output.DICT
)
# Access specific fields
words = data_dict['text']
confidences = data_dict['conf']
Structured Output Formats
# HOCR format (HTML with coordinates)
hocr = pytesseract.image_to_pdf_or_hocr(
Image.open('document.png'), extension='hocr'
)
# ALTO XML format
alto = pytesseract.image_to_alto_xml(Image.open('document.png'))
# TSV format
tsv = pytesseract.image_to_data(
Image.open('document.png'), output_type=pytesseract.Output.STRING
)
Searchable PDF Output
# Create searchable PDF from image
pdf_bytes = pytesseract.image_to_pdf_or_hocr(
Image.open('document.png'), extension='pdf'
)
with open('searchable.pdf', 'wb') as f:
f.write(pdf_bytes)
# Multiple images to single PDF
from PIL import Image
import io
images = [Image.open(f'page_{i}.png') for i in range(1, 4)]
pdf_pages = []
for img in images:
pdf_pages.append(pytesseract.image_to_pdf_or_hocr(img, extension='pdf'))
# Combine using PyPDF2 or similar
OSD (Orientation and Script Detection)
# Detect orientation and script
osd = pytesseract.image_to_osd(Image.open('document.png'))
print(osd)
# Parse OSD output
osd_dict = pytesseract.image_to_osd(
Image.open('document.png'), output_type=pytesseract.Output.DICT
)
rotation = osd_dict['rotate']
script = osd_dict['script']
Language and Configuration Options
Language Selection
# Single language
text = pytesseract.image_to_string(image, lang='fra') # French
# Multiple languages
text = pytesseract.image_to_string(image, lang='eng+fra+deu')
# List available languages
print(pytesseract.get_languages())
Page Segmentation Modes (PSM)
# Configure PSM via config string
config = '--psm 6' # Assume uniform block of text
text = pytesseract.image_to_string(image, config=config)
| PSM | Mode | Use Case |
|---|---|---|
| 0 | OSD only | Orientation and script detection |
| 1 | Auto + OSD | Automatic with OSD |
| 3 | Auto (default) | Fully automatic |
| 4 | Single column | Variable text sizes |
| 6 | Single block | Uniform text block |
| 7 | Single line | Single text line |
| 8 | Single word | Single word |
| 10 | Single character | Single character |
| 11 | Sparse text | Find text anywhere |
| 12 | Sparse text + OSD | Sparse with OSD |
| 13 | Raw line | Treat as single line, no preprocessing |
OCR Engine Modes (OEM)
# Configure OEM
config = '--oem 1' # LSTM only (neural network)
text = pytesseract.image_to_string(image, config=config)
| OEM | Mode | Description |
|---|---|---|
| 0 | Legacy | Original Tesseract engine |
| 1 | LSTM | Neural network engine |
| 2 | Legacy + LSTM | Combined engines |
| 3 | Default | Best available |
Character Whitelisting/Blacklisting
# Only allow specific characters (digits only)
config = '-c tessedit_char_whitelist=0123456789'
text = pytesseract.image_to_string(image, config=config)
# Exclude specific characters
config = '-c tessedit_char_blacklist=@#$%'
text = pytesseract.image_to_string(image, config=config)
# Alphanumeric only
config = '-c tessedit_char_whitelist=ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789'
Combined Configuration
# Multiple configuration options
config = '--psm 6 --oem 1 -c tessedit_char_whitelist=0123456789'
text = pytesseract.image_to_string(image, lang='eng', config=config)
Handling Different Image Formats
Supported Formats
from PIL import Image
# Common formats work directly
formats = ['png', 'jpg', 'jpeg', 'tiff', 'bmp', 'gif', 'webp']
# Load any supported format
image = Image.open('document.tiff')
text = pytesseract.image_to_string(image)
PDF Processing
from pdf2image import convert_from_path
# Convert PDF pages to images
pages = convert_from_path('document.pdf', dpi=300)
# Extract text from each page
all_text = []
for i, page in enumerate(pages):
text = pytesseract.image_to_string(page)
all_text.append(f"--- Page {i + 1} ---\n{text}")
full_text = '\n'.join(all_text)
Multi-page TIFF
from PIL import Image
# Open multi-page TIFF
tiff = Image.open('multipage.tiff')
all_text = []
try:
while True:
text = pytesseract.image_to_string(tiff)
all_text.append(text)
tiff.seek(tiff.tell() + 1)
except EOFError:
pass # End of pages
full_text = '\n'.join(all_text)
Screenshots and Clipboard
import pyautogui
from PIL import ImageGrab
# From screenshot
screenshot = pyautogui.screenshot()
text = pytesseract.image_to_string(screenshot)
# From clipboard
clipboard_image = ImageGrab.grabclipboard()
if clipboard_image:
text = pytesseract.image_to_string(clipboard_image)
URL Images
import requests
from PIL import Image
from io import BytesIO
# Download and process
response = requests.get('https://example.com/image.png')
image = Image.open(BytesIO(response.content))
text = pytesseract.image_to_string(image)
Post-processing Extracted Text
Basic Cleaning
import re
def clean_text(text):
"""Clean OCR output for better usability."""
# Remove extra whitespace
text = ' '.join(text.split())
# Fix common OCR errors
replacements = {
'|': 'I',
'0': 'O', # Context-dependent
'l': '1', # Context-dependent
'\n\n': '\n',
}
for old, new in replacements.items():
text = text.replace(old, new)
return text.strip()
Pattern-based Extraction
import re
def extract_emails(text):
"""Extract email addresses from OCR text."""
pattern = r'[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,}'
return re.findall(pattern, text)
def extract_phone_numbers(text):
"""Extract UK phone numbers."""
pattern = r'(?:(?:\+44\s?|0)(?:\d\s?){9,10})'
return re.findall(pattern, text)
def extract_dates(text):
"""Extract dates in common formats."""
patterns = [
r'\d{1,2}/\d{1,2}/\d{2,4}', # DD/MM/YYYY
r'\d{1,2}-\d{1,2}-\d{2,4}', # DD-MM-YYYY
r'\d{1,2}\s+\w+\s+\d{4}', # 15 January 2024
]
dates = []
for pattern in patterns:
dates.extend(re.findall(pattern, text))
return dates
Confidence Filtering
def extract_high_confidence_text(image, min_confidence=60):
"""Extract only text with confidence above threshold."""
data = pytesseract.image_to_data(
image, output_type=pytesseract.Output.DICT
)
words = []
for i, conf in enumerate(data['conf']):
if int(conf) >= min_confidence and data['text'][i].strip():
words.append(data['text'][i])
return ' '.join(words)
Spell Checking
Requires pyspellchecker (uv add pyspellchecker); it imports as spellchecker.
from spellchecker import SpellChecker
def correct_spelling(text):
"""Correct common OCR spelling errors."""
spell = SpellChecker()
words = text.split()
corrected = []
for word in words:
# Skip numbers and short words
if word.isdigit() or len(word) <= 2:
corrected.append(word)
continue
correction = spell.correction(word.lower())
if correction and correction != word.lower():
# Preserve original case
if word.isupper():
corrected.append(correction.upper())
elif word[0].isupper():
corrected.append(correction.capitalize())
else:
corrected.append(correction)
else:
corrected.append(word)
return ' '.join(corrected)
Line Reconstruction
def reconstruct_lines(image):
"""Reconstruct text preserving line structure."""
data = pytesseract.image_to_data(
image, output_type=pytesseract.Output.DICT
)
lines = {}
for i, text in enumerate(data['text']):
if text.strip():
line_num = data['line_num'][i]
block_num = data['block_num'][i]
key = (block_num, line_num)
if key not in lines:
lines[key] = []
lines[key].append(text)
# Join words in each line
result = []
for key in sorted(lines.keys()):
result.append(' '.join(lines[key]))
return '\n'.join(result)
Common Use Cases
Document Scanning
def scan_document(image_path, output_format='text'):
"""Complete document scanning workflow."""
# Preprocess
image = preprocess_image(image_path)
# Detect orientation and correct
try:
osd = pytesseract.image_to_osd(image, output_type=pytesseract.Output.DICT)
if osd['rotate'] != 0:
image = rotate_image(image, osd['rotate'])
except pytesseract.TesseractError:
pass # Skip if OSD fails
# Extract based on format
if output_format == 'text':
return pytesseract.image_to_string(image, lang='eng', config='--psm 3')
elif output_format == 'pdf':
return pytesseract.image_to_pdf_or_hocr(image, extension='pdf')
elif output_format == 'hocr':
return pytesseract.image_to_pdf_or_hocr(image, extension='hocr')
Invoice Data Extraction
def extract_invoice_data(image):
"""Extract structured data from invoice image."""
text = pytesseract.image_to_string(image)
data = {
'invoice_number': None,
'date': None,
'total': None,
'items': []
}
# Extract invoice number
inv_match = re.search(r'Invoice\s*#?\s*:?\s*(\w+)', text, re.I)
if inv_match:
data['invoice_number'] = inv_match.group(1)
# Extract date
date_match = re.search(r'Date\s*:?\s*(\d{1,2}[/-]\d{1,2}[/-]\d{2,4})', text)
if date_match:
data['date'] = date_match.group(1)
# Extract total
total_match = re.search(r'Total\s*:?\s*[£$]?\s*([\d,]+\.?\d*)', text, re.I)
if total_match:
data['total'] = total_match.group(1)
return data
Receipt Processing
def process_receipt(image_path):
"""Extract data from receipt images."""
# Receipts often need specific preprocessing
image = cv2.imread(image_path)
grey = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)
# Increase contrast for thermal receipts
clahe = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(8, 8))
enhanced = clahe.apply(grey)
# Use PSM 4 for column-like text
config = '--psm 4 --oem 1'
text = pytesseract.image_to_string(enhanced, config=config)
# Parse line items
lines = text.split('\n')
items = []
for line in lines:
# Match pattern: item name followed by price
match = re.search(r'^(.+?)\s+[£$]?([\d.]+)$', line.strip())
if match:
items.append({
'name': match.group(1).strip(),
'price': float(match.group(2))
})
return items
Business Card Reader
def read_business_card(image):
"""Extract contact information from business card."""
text = pytesseract.image_to_string(image, config='--psm 11')
contact = {
'name': None,
'email': None,
'phone': None,
'company': None
}
# Extract email
email_match = re.search(r'[\w.+-]+@[\w-]+\.[\w.-]+', text)
if email_match:
contact['email'] = email_match.group()
# Extract phone
phone_match = re.search(r'(?:\+\d{1,3}[-.\s]?)?\(?\d{3}\)?[-.\s]?\d{3}[-.\s]?\d{4}', text)
if phone_match:
contact['phone'] = phone_match.group()
return contact
License Plate Recognition
def read_license_plate(image):
"""Extract text from license plate image."""
# Preprocess for plates
grey = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)
# Apply bilateral filter to reduce noise while keeping edges
filtered = cv2.bilateralFilter(grey, 11, 17, 17)
# Edge detection
edged = cv2.Canny(filtered, 30, 200)
# Find contours and locate plate region
# ... (contour detection code)
# OCR with alphanumeric whitelist
config = '--psm 8 -c tessedit_char_whitelist=ABCDEFGHIJKLMNOPQRSTUVWXYZ0123456789'
plate_text = pytesseract.image_to_string(plate_region, config=config)
return plate_text.strip().replace(' ', '')
Table Data Extraction
def extract_table(image):
"""Extract tabular data from image."""
data = pytesseract.image_to_data(
image, output_type=pytesseract.Output.DICT
)
# Group text by rows based on vertical position
rows = {}
for i, text in enumerate(data['text']):
if text.strip():
# Use top coordinate as row key (with tolerance)
row_key = data['top'][i] // 10 * 10
if row_key not in rows:
rows[row_key] = []
rows[row_key].append({
'text': text,
'left': data['left'][i]
})
# Sort cells within each row by x position
table = []
for row_key in sorted(rows.keys()):
row = sorted(rows[row_key], key=lambda x: x['left'])
table.append([cell['text'] for cell in row])
return table
Performance Optimisation Strategies
Batch Processing
from concurrent.futures import ThreadPoolExecutor, as_completed
from pathlib import Path
def process_batch(image_paths, max_workers=4):
"""Process multiple images in parallel."""
results = {}
with ThreadPoolExecutor(max_workers=max_workers) as executor:
future_to_path = {
executor.submit(pytesseract.image_to_string, Image.open(path)): path
for path in image_paths
}
for future in as_completed(future_to_path):
path = future_to_path[future]
try:
results[path] = future.result()
except Exception as e:
results[path] = f"Error: {e}"
return results
Memory-efficient Processing
def process_large_images(image_paths):
"""Process images with minimal memory usage."""
for path in image_paths:
# Load, process, and immediately release
with Image.open(path) as img:
text = pytesseract.image_to_string(img)
yield path, text
# Image is automatically closed and memory released
Region of Interest (ROI) Processing
def process_roi(image, regions):
"""Process only specific regions of an image."""
results = {}
for name, (x, y, w, h) in regions.items():
# Crop to region of interest
if isinstance(image, np.ndarray):
roi = image[y:y+h, x:x+w]
else:
roi = image.crop((x, y, x+w, y+h))
results[name] = pytesseract.image_to_string(roi)
return results
# Example usage
regions = {
'header': (0, 0, 800, 100),
'body': (0, 100, 800, 600),
'footer': (0, 600, 800, 100)
}
result = process_roi(image, regions)
Caching Results
import hashlib
import json
from pathlib import Path
class OCRCache:
def __init__(self, cache_dir='.ocr_cache'):
self.cache_dir = Path(cache_dir)
self.cache_dir.mkdir(exist_ok=True)
def _get_hash(self, image_path):
with open(image_path, 'rb') as f:
return hashlib.md5(f.read()).hexdigest()
def get_or_extract(self, image_path, **kwargs):
cache_key = self._get_hash(image_path)
cache_file = self.cache_dir / f"{cache_key}.json"
if cache_file.exists():
with open(cache_file) as f:
return json.load(f)['text']
# Extract and cache
text = pytesseract.image_to_string(image_path, **kwargs)
with open(cache_file, 'w') as f:
json.dump({'text': text}, f)
return text
Optimising for Speed vs Accuracy
# Fast mode - lower accuracy
fast_config = '--psm 6 --oem 1'
text = pytesseract.image_to_string(image, config=fast_config)
# Accurate mode - slower but more precise
accurate_config = '--psm 3 --oem 1'
text = pytesseract.image_to_string(image, config=accurate_config)
# Custom trained data for specific fonts
config = '--tessdata-dir /path/to/custom/traineddata --psm 6'
Resolution Optimisation
def optimise_for_ocr(image, target_dpi=300):
"""Resize image to optimal DPI for OCR."""
# Get current dimensions
if hasattr(image, 'info') and 'dpi' in image.info:
current_dpi = image.info['dpi'][0]
else:
current_dpi = 72 # Assume screen resolution
if current_dpi < target_dpi:
scale = target_dpi / current_dpi
new_size = (int(image.width * scale), int(image.height * scale))
return image.resize(new_size, Image.LANCZOS)
return image
Troubleshooting Common OCR Issues
Diagnostic Function
def diagnose_ocr(image_path):
"""Run diagnostics on image for OCR issues."""
image = Image.open(image_path)
print(f"Image size: {image.size}")
print(f"Image mode: {image.mode}")
print(f"Image format: {image.format}")
# Check resolution
if hasattr(image, 'info') and 'dpi' in image.info:
print(f"DPI: {image.info['dpi']}")
# Try OSD
try:
osd = pytesseract.image_to_osd(image)
print(f"OSD: {osd}")
except pytesseract.TesseractError as e:
print(f"OSD failed: {e}")
# Get confidence data
data = pytesseract.image_to_data(
image, output_type=pytesseract.Output.DICT
)
confidences = [int(c) for c in data['conf'] if int(c) > 0]
if confidences:
avg_conf = sum(confidences) / len(confidences)
print(f"Average confidence: {avg_conf:.1f}%")
Common Issues and Solutions
| Issue | Cause | Solution |
|---|---|---|
| Empty output | Low resolution | Scale image to 300+ DPI |
| Garbled text | Wrong language | Set correct lang parameter |
| Missing characters | Poor contrast | Apply thresholding |
| Extra characters | Image noise | Apply denoising filters |
| Rotated output | Skewed image | Use deskewing or OSD |
| Slow processing | Large images | Crop to ROI or reduce resolution |
| Special chars missing | Character whitelist | Remove or adjust whitelist |
Handling Specific Errors
from PIL import Image
import pytesseract
def safe_ocr(image_path, **kwargs):
"""OCR with comprehensive error handling."""
try:
image = Image.open(image_path)
# Check for minimum size
if image.width < 10 or image.height < 10:
return {'error': 'Image too small'}
# Convert if necessary
if image.mode not in ('L', 'RGB'):
image = image.convert('RGB')
text = pytesseract.image_to_string(image, **kwargs)
return {'text': text, 'success': True}
except pytesseract.TesseractNotFoundError:
return {'error': 'Tesseract not installed or not in PATH'}
except pytesseract.TesseractError as e:
return {'error': f'Tesseract error: {e}'}
except IOError as e:
return {'error': f'Cannot open image: {e}'}
except Exception as e:
return {'error': f'Unexpected error: {e}'}
Improving Poor Results
def improve_ocr_results(image_path):
"""Try multiple strategies to improve OCR results."""
image = cv2.imread(image_path)
grey = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)
strategies = [
('Original', grey),
('Otsu threshold', cv2.threshold(grey, 0, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU)[1]),
('Adaptive threshold', cv2.adaptiveThreshold(grey, 255, cv2.ADAPTIVE_THRESH_GAUSSIAN_C, cv2.THRESH_BINARY, 11, 2)),
('Inverted', cv2.bitwise_not(grey)),
('Scaled 2x', cv2.resize(grey, None, fx=2, fy=2, interpolation=cv2.INTER_CUBIC)),
]
best_result = ''
best_confidence = 0
for name, processed in strategies:
data = pytesseract.image_to_data(
processed, output_type=pytesseract.Output.DICT
)
# Calculate average confidence
confs = [int(c) for c in data['conf'] if int(c) > 0]
avg_conf = sum(confs) / len(confs) if confs else 0
if avg_conf > best_confidence:
best_confidence = avg_conf
best_result = ' '.join([t for t in data['text'] if t.strip()])
print(f"{name}: {avg_conf:.1f}% confidence")
return best_result
Quick Reference
Essential Commands
# Basic extraction
text = pytesseract.image_to_string(image)
# With language
text = pytesseract.image_to_string(image, lang='eng')
# With configuration
text = pytesseract.image_to_string(image, config='--psm 6 --oem 1')
# Get bounding boxes
boxes = pytesseract.image_to_boxes(image)
# Get detailed data
data = pytesseract.image_to_data(image, output_type=pytesseract.Output.DICT)
# Create searchable PDF
pdf = pytesseract.image_to_pdf_or_hocr(image, extension='pdf')
# Detect orientation
osd = pytesseract.image_to_osd(image)
Common Configuration Strings
# Digits only
'--psm 6 -c tessedit_char_whitelist=0123456789'
# Single line
'--psm 7'
# Sparse text
'--psm 11'
# LSTM engine only
'--oem 1'
# Multiple options
'--psm 6 --oem 1 -c preserve_interword_spaces=1'
Preprocessing Quick Reference
# Grayscale
grey = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)
# Threshold
_, binary = cv2.threshold(grey, 0, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU)
# Denoise
denoised = cv2.fastNlMeansDenoising(grey, None, 10, 7, 21)
# Scale
scaled = cv2.resize(image, None, fx=2, fy=2, interpolation=cv2.INTER_CUBIC)
Related Topics
- OpenCV - Advanced image preprocessing and computer vision
- Pillow - Python imaging library for image manipulation
- pdf2image - PDF to image conversion for document processing
- spaCy/NLTK - Natural language processing for text post-processing
- pandas - Data manipulation for structured OCR output
- asyncio - Asynchronous processing for high-volume OCR tasks