Available for day contractsFrom 21st September I have availability for day and half day contracts. Please contact for more information.

Contact →
mikepreston.org

Python pytesseract

A Python wrapper for Google's Tesseract-OCR engine, enabling text extraction from images.

Python pytesseract

A Python wrapper for Google's Tesseract-OCR engine, enabling text extraction from images.

Overview

pytesseract provides a simple interface to Tesseract OCR, allowing Python applications to extract text from images, PDFs, and scanned documents. It supports multiple languages, various output formats, and integrates seamlessly with image processing libraries like Pillow and OpenCV.

Output FormatsInput ImagePreprocessingpytesseractTesseract OCRText OutputPlain TextBounding BoxesSearchable PDFHOCR/XMLOutput FormatsInput ImagePreprocessingpytesseractTesseract OCRText OutputPlain TextBounding BoxesSearchable PDFHOCR/XML

Installation and Setup

Installing Tesseract OCR Engine

# Ubuntu/Debian
sudo apt-get install tesseract-ocr

# macOS
brew install tesseract

# Windows - download installer from GitHub releases
# https://github.com/UB-Mannheim/tesseract/wiki

# Install additional language packs
sudo apt-get install tesseract-ocr-fra tesseract-ocr-deu  # French, German

Installing pytesseract

uv add pytesseract pillow opencv-python

Configuration

import pytesseract
from PIL import Image

# Set Tesseract path (required on Windows, optional on Unix)
pytesseract.pytesseract.tesseract_cmd = r'C:\Program Files\Tesseract-OCR\tesseract.exe'

# Verify installation
print(pytesseract.get_tesseract_version())

Image Preprocessing Techniques

Proper preprocessing dramatically improves OCR accuracy. Apply these techniques before extraction.

Original ImageGrayscale ConversionNoise ReductionThresholdingDeskewingScalingReady for OCROriginal ImageGrayscale ConversionNoise ReductionThresholdingDeskewingScalingReady for OCR

Grayscale Conversion

import cv2
from PIL import Image
import numpy as np

# Using OpenCV
image = cv2.imread('document.png')
grey = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)

# Using Pillow
image = Image.open('document.png').convert('L')

Noise Reduction

# Gaussian blur for general noise
denoised = cv2.GaussianBlur(grey, (5, 5), 0)

# Median blur for salt-and-pepper noise
denoised = cv2.medianBlur(grey, 3)

# Non-local means denoising (best quality, slower)
denoised = cv2.fastNlMeansDenoising(grey, None, 10, 7, 21)

Thresholding

# Simple binary threshold
_, binary = cv2.threshold(grey, 127, 255, cv2.THRESH_BINARY)

# Otsu's automatic thresholding (recommended)
_, binary = cv2.threshold(grey, 0, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU)

# Adaptive thresholding for uneven lighting
adaptive = cv2.adaptiveThreshold(
    grey, 255, cv2.ADAPTIVE_THRESH_GAUSSIAN_C, cv2.THRESH_BINARY, 11, 2
)

Deskewing

def deskew(image):
    """Correct image rotation using Hough transform."""
    coords = np.column_stack(np.where(image > 0))
    angle = cv2.minAreaRect(coords)[-1]

    if angle < -45:
        angle = -(90 + angle)
    else:
        angle = -angle

    (h, w) = image.shape[:2]
    centre = (w // 2, h // 2)
    M = cv2.getRotationMatrix2D(centre, angle, 1.0)
    rotated = cv2.warpAffine(
        image, M, (w, h), flags=cv2.INTER_CUBIC, borderMode=cv2.BORDER_REPLICATE
    )
    return rotated

Scaling for Small Text

# Scale up images with small text (optimal DPI is 300)
def scale_image(image, scale_factor=2):
    width = int(image.shape[1] * scale_factor)
    height = int(image.shape[0] * scale_factor)
    return cv2.resize(image, (width, height), interpolation=cv2.INTER_CUBIC)

Complete Preprocessing Pipeline

def preprocess_image(image_path):
    """Apply full preprocessing pipeline for optimal OCR."""
    # Load image
    image = cv2.imread(image_path)

    # Convert to grayscale
    grey = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)

    # Remove noise
    denoised = cv2.fastNlMeansDenoising(grey, None, 10, 7, 21)

    # Apply adaptive thresholding
    binary = cv2.adaptiveThreshold(
        denoised, 255, cv2.ADAPTIVE_THRESH_GAUSSIAN_C, cv2.THRESH_BINARY, 11, 2
    )

    # Deskew if needed
    corrected = deskew(binary)

    return corrected

OCR Extraction Methods

Basic Text Extraction

import pytesseract
from PIL import Image

# Simple extraction
text = pytesseract.image_to_string(Image.open('document.png'))
print(text)

# From OpenCV image (numpy array)
text = pytesseract.image_to_string(cv2_image)

# From file path directly
text = pytesseract.image_to_string('document.png')

Bounding Box Data

# Get bounding boxes for each character
boxes = pytesseract.image_to_boxes(Image.open('document.png'))
for box in boxes.splitlines():
    char, x1, y1, x2, y2, page = box.split()
    print(f"Character '{char}' at ({x1}, {y1}) to ({x2}, {y2})")

# Get detailed data including word-level boxes
data = pytesseract.image_to_data(Image.open('document.png'))
print(data)

# As dictionary for easier processing
data_dict = pytesseract.image_to_data(
    Image.open('document.png'), output_type=pytesseract.Output.DICT
)

# Access specific fields
words = data_dict['text']
confidences = data_dict['conf']

Structured Output Formats

# HOCR format (HTML with coordinates)
hocr = pytesseract.image_to_pdf_or_hocr(
    Image.open('document.png'), extension='hocr'
)

# ALTO XML format
alto = pytesseract.image_to_alto_xml(Image.open('document.png'))

# TSV format
tsv = pytesseract.image_to_data(
    Image.open('document.png'), output_type=pytesseract.Output.STRING
)

Searchable PDF Output

# Create searchable PDF from image
pdf_bytes = pytesseract.image_to_pdf_or_hocr(
    Image.open('document.png'), extension='pdf'
)

with open('searchable.pdf', 'wb') as f:
    f.write(pdf_bytes)

# Multiple images to single PDF
from PIL import Image
import io

images = [Image.open(f'page_{i}.png') for i in range(1, 4)]
pdf_pages = []

for img in images:
    pdf_pages.append(pytesseract.image_to_pdf_or_hocr(img, extension='pdf'))

# Combine using PyPDF2 or similar

OSD (Orientation and Script Detection)

# Detect orientation and script
osd = pytesseract.image_to_osd(Image.open('document.png'))
print(osd)

# Parse OSD output
osd_dict = pytesseract.image_to_osd(
    Image.open('document.png'), output_type=pytesseract.Output.DICT
)
rotation = osd_dict['rotate']
script = osd_dict['script']

Language and Configuration Options

Language Selection

# Single language
text = pytesseract.image_to_string(image, lang='fra')  # French

# Multiple languages
text = pytesseract.image_to_string(image, lang='eng+fra+deu')

# List available languages
print(pytesseract.get_languages())

Page Segmentation Modes (PSM)

# Configure PSM via config string
config = '--psm 6'  # Assume uniform block of text
text = pytesseract.image_to_string(image, config=config)
PSM Mode Use Case
0 OSD only Orientation and script detection
1 Auto + OSD Automatic with OSD
3 Auto (default) Fully automatic
4 Single column Variable text sizes
6 Single block Uniform text block
7 Single line Single text line
8 Single word Single word
10 Single character Single character
11 Sparse text Find text anywhere
12 Sparse text + OSD Sparse with OSD
13 Raw line Treat as single line, no preprocessing

OCR Engine Modes (OEM)

# Configure OEM
config = '--oem 1'  # LSTM only (neural network)
text = pytesseract.image_to_string(image, config=config)
OEM Mode Description
0 Legacy Original Tesseract engine
1 LSTM Neural network engine
2 Legacy + LSTM Combined engines
3 Default Best available

Character Whitelisting/Blacklisting

# Only allow specific characters (digits only)
config = '-c tessedit_char_whitelist=0123456789'
text = pytesseract.image_to_string(image, config=config)

# Exclude specific characters
config = '-c tessedit_char_blacklist=@#$%'
text = pytesseract.image_to_string(image, config=config)

# Alphanumeric only
config = '-c tessedit_char_whitelist=ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789'

Combined Configuration

# Multiple configuration options
config = '--psm 6 --oem 1 -c tessedit_char_whitelist=0123456789'
text = pytesseract.image_to_string(image, lang='eng', config=config)

Handling Different Image Formats

Supported Formats

from PIL import Image

# Common formats work directly
formats = ['png', 'jpg', 'jpeg', 'tiff', 'bmp', 'gif', 'webp']

# Load any supported format
image = Image.open('document.tiff')
text = pytesseract.image_to_string(image)

PDF Processing

from pdf2image import convert_from_path

# Convert PDF pages to images
pages = convert_from_path('document.pdf', dpi=300)

# Extract text from each page
all_text = []
for i, page in enumerate(pages):
    text = pytesseract.image_to_string(page)
    all_text.append(f"--- Page {i + 1} ---\n{text}")

full_text = '\n'.join(all_text)

Multi-page TIFF

from PIL import Image

# Open multi-page TIFF
tiff = Image.open('multipage.tiff')
all_text = []

try:
    while True:
        text = pytesseract.image_to_string(tiff)
        all_text.append(text)
        tiff.seek(tiff.tell() + 1)
except EOFError:
    pass  # End of pages

full_text = '\n'.join(all_text)

Screenshots and Clipboard

import pyautogui
from PIL import ImageGrab

# From screenshot
screenshot = pyautogui.screenshot()
text = pytesseract.image_to_string(screenshot)

# From clipboard
clipboard_image = ImageGrab.grabclipboard()
if clipboard_image:
    text = pytesseract.image_to_string(clipboard_image)

URL Images

import requests
from PIL import Image
from io import BytesIO

# Download and process
response = requests.get('https://example.com/image.png')
image = Image.open(BytesIO(response.content))
text = pytesseract.image_to_string(image)

Post-processing Extracted Text

Basic Cleaning

import re

def clean_text(text):
    """Clean OCR output for better usability."""
    # Remove extra whitespace
    text = ' '.join(text.split())

    # Fix common OCR errors
    replacements = {
        '|': 'I',
        '0': 'O',  # Context-dependent
        'l': '1',  # Context-dependent
        '\n\n': '\n',
    }

    for old, new in replacements.items():
        text = text.replace(old, new)

    return text.strip()

Pattern-based Extraction

import re

def extract_emails(text):
    """Extract email addresses from OCR text."""
    pattern = r'[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,}'
    return re.findall(pattern, text)

def extract_phone_numbers(text):
    """Extract UK phone numbers."""
    pattern = r'(?:(?:\+44\s?|0)(?:\d\s?){9,10})'
    return re.findall(pattern, text)

def extract_dates(text):
    """Extract dates in common formats."""
    patterns = [
        r'\d{1,2}/\d{1,2}/\d{2,4}',  # DD/MM/YYYY
        r'\d{1,2}-\d{1,2}-\d{2,4}',  # DD-MM-YYYY
        r'\d{1,2}\s+\w+\s+\d{4}',    # 15 January 2024
    ]
    dates = []
    for pattern in patterns:
        dates.extend(re.findall(pattern, text))
    return dates

Confidence Filtering

def extract_high_confidence_text(image, min_confidence=60):
    """Extract only text with confidence above threshold."""
    data = pytesseract.image_to_data(
        image, output_type=pytesseract.Output.DICT
    )

    words = []
    for i, conf in enumerate(data['conf']):
        if int(conf) >= min_confidence and data['text'][i].strip():
            words.append(data['text'][i])

    return ' '.join(words)

Spell Checking

Requires pyspellchecker (uv add pyspellchecker); it imports as spellchecker.

from spellchecker import SpellChecker

def correct_spelling(text):
    """Correct common OCR spelling errors."""
    spell = SpellChecker()
    words = text.split()
    corrected = []

    for word in words:
        # Skip numbers and short words
        if word.isdigit() or len(word) <= 2:
            corrected.append(word)
            continue

        correction = spell.correction(word.lower())
        if correction and correction != word.lower():
            # Preserve original case
            if word.isupper():
                corrected.append(correction.upper())
            elif word[0].isupper():
                corrected.append(correction.capitalize())
            else:
                corrected.append(correction)
        else:
            corrected.append(word)

    return ' '.join(corrected)

Line Reconstruction

def reconstruct_lines(image):
    """Reconstruct text preserving line structure."""
    data = pytesseract.image_to_data(
        image, output_type=pytesseract.Output.DICT
    )

    lines = {}
    for i, text in enumerate(data['text']):
        if text.strip():
            line_num = data['line_num'][i]
            block_num = data['block_num'][i]
            key = (block_num, line_num)

            if key not in lines:
                lines[key] = []
            lines[key].append(text)

    # Join words in each line
    result = []
    for key in sorted(lines.keys()):
        result.append(' '.join(lines[key]))

    return '\n'.join(result)

Common Use Cases

Document Scanning

def scan_document(image_path, output_format='text'):
    """Complete document scanning workflow."""
    # Preprocess
    image = preprocess_image(image_path)

    # Detect orientation and correct
    try:
        osd = pytesseract.image_to_osd(image, output_type=pytesseract.Output.DICT)
        if osd['rotate'] != 0:
            image = rotate_image(image, osd['rotate'])
    except pytesseract.TesseractError:
        pass  # Skip if OSD fails

    # Extract based on format
    if output_format == 'text':
        return pytesseract.image_to_string(image, lang='eng', config='--psm 3')
    elif output_format == 'pdf':
        return pytesseract.image_to_pdf_or_hocr(image, extension='pdf')
    elif output_format == 'hocr':
        return pytesseract.image_to_pdf_or_hocr(image, extension='hocr')

Invoice Data Extraction

def extract_invoice_data(image):
    """Extract structured data from invoice image."""
    text = pytesseract.image_to_string(image)

    data = {
        'invoice_number': None,
        'date': None,
        'total': None,
        'items': []
    }

    # Extract invoice number
    inv_match = re.search(r'Invoice\s*#?\s*:?\s*(\w+)', text, re.I)
    if inv_match:
        data['invoice_number'] = inv_match.group(1)

    # Extract date
    date_match = re.search(r'Date\s*:?\s*(\d{1,2}[/-]\d{1,2}[/-]\d{2,4})', text)
    if date_match:
        data['date'] = date_match.group(1)

    # Extract total
    total_match = re.search(r'Total\s*:?\s*[£$]?\s*([\d,]+\.?\d*)', text, re.I)
    if total_match:
        data['total'] = total_match.group(1)

    return data

Receipt Processing

def process_receipt(image_path):
    """Extract data from receipt images."""
    # Receipts often need specific preprocessing
    image = cv2.imread(image_path)
    grey = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)

    # Increase contrast for thermal receipts
    clahe = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(8, 8))
    enhanced = clahe.apply(grey)

    # Use PSM 4 for column-like text
    config = '--psm 4 --oem 1'
    text = pytesseract.image_to_string(enhanced, config=config)

    # Parse line items
    lines = text.split('\n')
    items = []
    for line in lines:
        # Match pattern: item name followed by price
        match = re.search(r'^(.+?)\s+[£$]?([\d.]+)$', line.strip())
        if match:
            items.append({
                'name': match.group(1).strip(),
                'price': float(match.group(2))
            })

    return items

Business Card Reader

def read_business_card(image):
    """Extract contact information from business card."""
    text = pytesseract.image_to_string(image, config='--psm 11')

    contact = {
        'name': None,
        'email': None,
        'phone': None,
        'company': None
    }

    # Extract email
    email_match = re.search(r'[\w.+-]+@[\w-]+\.[\w.-]+', text)
    if email_match:
        contact['email'] = email_match.group()

    # Extract phone
    phone_match = re.search(r'(?:\+\d{1,3}[-.\s]?)?\(?\d{3}\)?[-.\s]?\d{3}[-.\s]?\d{4}', text)
    if phone_match:
        contact['phone'] = phone_match.group()

    return contact

License Plate Recognition

def read_license_plate(image):
    """Extract text from license plate image."""
    # Preprocess for plates
    grey = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)

    # Apply bilateral filter to reduce noise while keeping edges
    filtered = cv2.bilateralFilter(grey, 11, 17, 17)

    # Edge detection
    edged = cv2.Canny(filtered, 30, 200)

    # Find contours and locate plate region
    # ... (contour detection code)

    # OCR with alphanumeric whitelist
    config = '--psm 8 -c tessedit_char_whitelist=ABCDEFGHIJKLMNOPQRSTUVWXYZ0123456789'
    plate_text = pytesseract.image_to_string(plate_region, config=config)

    return plate_text.strip().replace(' ', '')

Table Data Extraction

def extract_table(image):
    """Extract tabular data from image."""
    data = pytesseract.image_to_data(
        image, output_type=pytesseract.Output.DICT
    )

    # Group text by rows based on vertical position
    rows = {}
    for i, text in enumerate(data['text']):
        if text.strip():
            # Use top coordinate as row key (with tolerance)
            row_key = data['top'][i] // 10 * 10
            if row_key not in rows:
                rows[row_key] = []
            rows[row_key].append({
                'text': text,
                'left': data['left'][i]
            })

    # Sort cells within each row by x position
    table = []
    for row_key in sorted(rows.keys()):
        row = sorted(rows[row_key], key=lambda x: x['left'])
        table.append([cell['text'] for cell in row])

    return table

Performance Optimisation Strategies

Batch Processing

from concurrent.futures import ThreadPoolExecutor, as_completed
from pathlib import Path

def process_batch(image_paths, max_workers=4):
    """Process multiple images in parallel."""
    results = {}

    with ThreadPoolExecutor(max_workers=max_workers) as executor:
        future_to_path = {
            executor.submit(pytesseract.image_to_string, Image.open(path)): path
            for path in image_paths
        }

        for future in as_completed(future_to_path):
            path = future_to_path[future]
            try:
                results[path] = future.result()
            except Exception as e:
                results[path] = f"Error: {e}"

    return results

Memory-efficient Processing

def process_large_images(image_paths):
    """Process images with minimal memory usage."""
    for path in image_paths:
        # Load, process, and immediately release
        with Image.open(path) as img:
            text = pytesseract.image_to_string(img)
            yield path, text
        # Image is automatically closed and memory released

Region of Interest (ROI) Processing

def process_roi(image, regions):
    """Process only specific regions of an image."""
    results = {}

    for name, (x, y, w, h) in regions.items():
        # Crop to region of interest
        if isinstance(image, np.ndarray):
            roi = image[y:y+h, x:x+w]
        else:
            roi = image.crop((x, y, x+w, y+h))

        results[name] = pytesseract.image_to_string(roi)

    return results

# Example usage
regions = {
    'header': (0, 0, 800, 100),
    'body': (0, 100, 800, 600),
    'footer': (0, 600, 800, 100)
}
result = process_roi(image, regions)

Caching Results

import hashlib
import json
from pathlib import Path

class OCRCache:
    def __init__(self, cache_dir='.ocr_cache'):
        self.cache_dir = Path(cache_dir)
        self.cache_dir.mkdir(exist_ok=True)

    def _get_hash(self, image_path):
        with open(image_path, 'rb') as f:
            return hashlib.md5(f.read()).hexdigest()

    def get_or_extract(self, image_path, **kwargs):
        cache_key = self._get_hash(image_path)
        cache_file = self.cache_dir / f"{cache_key}.json"

        if cache_file.exists():
            with open(cache_file) as f:
                return json.load(f)['text']

        # Extract and cache
        text = pytesseract.image_to_string(image_path, **kwargs)
        with open(cache_file, 'w') as f:
            json.dump({'text': text}, f)

        return text

Optimising for Speed vs Accuracy

# Fast mode - lower accuracy
fast_config = '--psm 6 --oem 1'
text = pytesseract.image_to_string(image, config=fast_config)

# Accurate mode - slower but more precise
accurate_config = '--psm 3 --oem 1'
text = pytesseract.image_to_string(image, config=accurate_config)

# Custom trained data for specific fonts
config = '--tessdata-dir /path/to/custom/traineddata --psm 6'

Resolution Optimisation

def optimise_for_ocr(image, target_dpi=300):
    """Resize image to optimal DPI for OCR."""
    # Get current dimensions
    if hasattr(image, 'info') and 'dpi' in image.info:
        current_dpi = image.info['dpi'][0]
    else:
        current_dpi = 72  # Assume screen resolution

    if current_dpi < target_dpi:
        scale = target_dpi / current_dpi
        new_size = (int(image.width * scale), int(image.height * scale))
        return image.resize(new_size, Image.LANCZOS)

    return image

Troubleshooting Common OCR Issues

Diagnostic Function

def diagnose_ocr(image_path):
    """Run diagnostics on image for OCR issues."""
    image = Image.open(image_path)

    print(f"Image size: {image.size}")
    print(f"Image mode: {image.mode}")
    print(f"Image format: {image.format}")

    # Check resolution
    if hasattr(image, 'info') and 'dpi' in image.info:
        print(f"DPI: {image.info['dpi']}")

    # Try OSD
    try:
        osd = pytesseract.image_to_osd(image)
        print(f"OSD: {osd}")
    except pytesseract.TesseractError as e:
        print(f"OSD failed: {e}")

    # Get confidence data
    data = pytesseract.image_to_data(
        image, output_type=pytesseract.Output.DICT
    )
    confidences = [int(c) for c in data['conf'] if int(c) > 0]
    if confidences:
        avg_conf = sum(confidences) / len(confidences)
        print(f"Average confidence: {avg_conf:.1f}%")

Common Issues and Solutions

Issue Cause Solution
Empty output Low resolution Scale image to 300+ DPI
Garbled text Wrong language Set correct lang parameter
Missing characters Poor contrast Apply thresholding
Extra characters Image noise Apply denoising filters
Rotated output Skewed image Use deskewing or OSD
Slow processing Large images Crop to ROI or reduce resolution
Special chars missing Character whitelist Remove or adjust whitelist

Handling Specific Errors

from PIL import Image
import pytesseract

def safe_ocr(image_path, **kwargs):
    """OCR with comprehensive error handling."""
    try:
        image = Image.open(image_path)

        # Check for minimum size
        if image.width < 10 or image.height < 10:
            return {'error': 'Image too small'}

        # Convert if necessary
        if image.mode not in ('L', 'RGB'):
            image = image.convert('RGB')

        text = pytesseract.image_to_string(image, **kwargs)
        return {'text': text, 'success': True}

    except pytesseract.TesseractNotFoundError:
        return {'error': 'Tesseract not installed or not in PATH'}

    except pytesseract.TesseractError as e:
        return {'error': f'Tesseract error: {e}'}

    except IOError as e:
        return {'error': f'Cannot open image: {e}'}

    except Exception as e:
        return {'error': f'Unexpected error: {e}'}

Improving Poor Results

def improve_ocr_results(image_path):
    """Try multiple strategies to improve OCR results."""
    image = cv2.imread(image_path)
    grey = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)

    strategies = [
        ('Original', grey),
        ('Otsu threshold', cv2.threshold(grey, 0, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU)[1]),
        ('Adaptive threshold', cv2.adaptiveThreshold(grey, 255, cv2.ADAPTIVE_THRESH_GAUSSIAN_C, cv2.THRESH_BINARY, 11, 2)),
        ('Inverted', cv2.bitwise_not(grey)),
        ('Scaled 2x', cv2.resize(grey, None, fx=2, fy=2, interpolation=cv2.INTER_CUBIC)),
    ]

    best_result = ''
    best_confidence = 0

    for name, processed in strategies:
        data = pytesseract.image_to_data(
            processed, output_type=pytesseract.Output.DICT
        )

        # Calculate average confidence
        confs = [int(c) for c in data['conf'] if int(c) > 0]
        avg_conf = sum(confs) / len(confs) if confs else 0

        if avg_conf > best_confidence:
            best_confidence = avg_conf
            best_result = ' '.join([t for t in data['text'] if t.strip()])
            print(f"{name}: {avg_conf:.1f}% confidence")

    return best_result

Quick Reference

Essential Commands

# Basic extraction
text = pytesseract.image_to_string(image)

# With language
text = pytesseract.image_to_string(image, lang='eng')

# With configuration
text = pytesseract.image_to_string(image, config='--psm 6 --oem 1')

# Get bounding boxes
boxes = pytesseract.image_to_boxes(image)

# Get detailed data
data = pytesseract.image_to_data(image, output_type=pytesseract.Output.DICT)

# Create searchable PDF
pdf = pytesseract.image_to_pdf_or_hocr(image, extension='pdf')

# Detect orientation
osd = pytesseract.image_to_osd(image)

Common Configuration Strings

# Digits only
'--psm 6 -c tessedit_char_whitelist=0123456789'

# Single line
'--psm 7'

# Sparse text
'--psm 11'

# LSTM engine only
'--oem 1'

# Multiple options
'--psm 6 --oem 1 -c preserve_interword_spaces=1'

Preprocessing Quick Reference

# Grayscale
grey = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)

# Threshold
_, binary = cv2.threshold(grey, 0, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU)

# Denoise
denoised = cv2.fastNlMeansDenoising(grey, None, 10, 7, 21)

# Scale
scaled = cv2.resize(image, None, fx=2, fy=2, interpolation=cv2.INTER_CUBIC)

Related Topics

  • OpenCV - Advanced image preprocessing and computer vision
  • Pillow - Python imaging library for image manipulation
  • pdf2image - PDF to image conversion for document processing
  • spaCy/NLTK - Natural language processing for text post-processing
  • pandas - Data manipulation for structured OCR output
  • asyncio - Asynchronous processing for high-volume OCR tasks