Files
notes-cleaner/pdf_cleaner.py
T
Esa Kataja 3d143ce5a2 feat: Add optimize command
- Add whitespace trimming functionality
- Add optional PNG optimization with optipng
- Add progress tracking and error handling
- Improve imports organization
- Add proper docstrings and return values
2024-12-03 21:10:38 +02:00

270 lines
8.9 KiB
Python
Executable File

#!/usr/bin/env -S uv run
import os
import shutil
import subprocess
from pathlib import Path
from typing import List, Tuple
import click
import cv2
import img2pdf
import numpy as np
from pdf2image import convert_from_path
def ensure_temp_dir():
"""Ensure temporary directory exists and return its path."""
temp_dir = "temp_processed_images"
os.makedirs(temp_dir, exist_ok=True)
return temp_dir
def get_temp_files():
"""Get list of temporary PNG files in order."""
temp_dir = ensure_temp_dir()
files = [f for f in os.listdir(temp_dir) if f.endswith('.png')]
files.sort() # Ensure correct page order
return [os.path.join(temp_dir, f) for f in files]
@click.group()
def cli():
"""PDF cleaning toolbox for musical scores."""
pass
@cli.command()
@click.argument('input_pdf', type=click.Path(exists=True))
def extract(input_pdf):
"""Extract pages from PDF to temporary directory."""
temp_dir = ensure_temp_dir()
# Convert PDF to images
print(f"Extracting pages from {input_pdf}...")
pages = convert_from_path(input_pdf, dpi=400)
# Save each page
for i, page in enumerate(pages):
output_path = os.path.join(temp_dir, f"page_{i:03d}.png")
page.save(output_path, "PNG", optimize=False)
print(f"Saved page {i+1}/{len(pages)}")
print(f"Extracted {len(pages)} pages to {temp_dir}/")
@cli.command()
def deskew():
"""Deskew all pages in temporary directory."""
temp_files = get_temp_files()
if not temp_files:
print("No pages found in temporary directory. Run 'extract' first.")
return
for file_path in temp_files:
print(f"Deskewing {os.path.basename(file_path)}...")
# Read image
image = cv2.imread(file_path)
# Convert to grayscale
gray = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)
# Apply threshold to get binary image
_, binary = cv2.threshold(gray, 127, 255, cv2.THRESH_BINARY)
# Create a rectangular kernel that's wider than it is tall
# This helps detect horizontal lines
kernel_length = np.array(binary).shape[1]//80
horizontal_kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (kernel_length, 1))
# Detect horizontal lines
horizontal_lines = cv2.erode(binary, horizontal_kernel, iterations=3)
horizontal_lines = cv2.dilate(horizontal_lines, horizontal_kernel, iterations=3)
# Use probabilistic Hough transform to detect line segments
line_segments = cv2.HoughLinesP(
cv2.bitwise_not(horizontal_lines),
rho=1,
theta=np.pi/180,
threshold=100,
minLineLength=binary.shape[1]//4, # Lines must be at least 1/4 of image width
maxLineGap=20
)
if line_segments is not None and len(line_segments) > 0:
# Calculate angles of detected line segments
angles = []
for line in line_segments:
x1, y1, x2, y2 = line[0]
if x2 - x1 == 0: # Avoid division by zero
continue
angle = np.degrees(np.arctan2(y2 - y1, x2 - x1))
# Only consider angles that are close to horizontal
if abs(angle) < 20:
angles.append(angle)
if angles:
# Use median angle to avoid outliers
median_angle = np.median(angles)
# Only rotate if the angle is significant but not too large
if 0.5 < abs(median_angle) < 20:
height, width = image.shape[:2]
center = (width/2, height/2)
rotation_matrix = cv2.getRotationMatrix2D(center, median_angle, 1.0)
rotated = cv2.warpAffine(image, rotation_matrix, (width, height),
flags=cv2.INTER_CUBIC,
borderMode=cv2.BORDER_REPLICATE)
# Save rotated image
cv2.imwrite(file_path, rotated)
print(f" Rotated by {median_angle:.2f} degrees")
else:
print(" No significant rotation needed")
else:
print(" No valid horizontal lines found")
else:
print(" No line segments detected")
def trim_whitespace(image_path: str) -> bool:
"""Remove white space from around the image.
Returns:
bool: True if successful, False otherwise
"""
try:
# Read the image
img = cv2.imread(image_path)
if img is None:
print(f"Warning: Could not read image {image_path}")
return False
# Convert to grayscale
gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)
# Threshold the image
_, thresh = cv2.threshold(gray, 250, 255, cv2.THRESH_BINARY_INV)
# Find non-zero points
coords = cv2.findNonZero(thresh)
if coords is None:
print(f"Warning: No content found in {image_path}")
return False
# Get bounding rectangle
x, y, w, h = cv2.boundingRect(coords)
# Add small padding (20 pixels)
padding = 20
height, width = img.shape[:2]
x = max(0, x - padding)
y = max(0, y - padding)
w = min(width - x, w + 2 * padding)
h = min(height - y, h + 2 * padding)
# Crop the image
cropped = img[y:y+h, x:x+w]
# Save the cropped image
cv2.imwrite(image_path, cropped)
return True
except Exception as e:
print(f"Error processing {image_path}: {str(e)}")
return False
def check_optipng_installed() -> bool:
"""Check if optipng is installed."""
try:
result = subprocess.run(['optipng', '-v'], capture_output=True, text=True)
return result.returncode == 0
except FileNotFoundError:
return False
def run_optipng(file_path: str) -> bool:
"""Run optipng on a file with error handling.
Returns:
bool: True if successful, False otherwise
"""
try:
result = subprocess.run(
['optipng', '-o7', file_path],
capture_output=True,
text=True,
check=True
)
return True
except subprocess.CalledProcessError as e:
print(f"Error optimizing {file_path}: {e.stderr}")
return False
except Exception as e:
print(f"Unexpected error optimizing {file_path}: {str(e)}")
return False
@cli.command()
def optimize():
"""Optimize images by trimming whitespace and optionally running optipng."""
temp_dir = ensure_temp_dir()
temp_files = get_temp_files()
if not temp_files:
print("No pages found in temporary directory. Run 'extract' first.")
return
total_files = len(temp_files)
successful_trims = 0
successful_opts = 0
print(f"Processing {total_files} images...")
# Trim whitespace
print("\nTrimming whitespace from images...")
for i, file_path in enumerate(temp_files, 1):
print(f"[{i}/{total_files}] Processing {os.path.basename(file_path)}...", end='', flush=True)
if trim_whitespace(file_path):
successful_trims += 1
print(" ")
else:
print(" ")
# Optimize with optipng if available
if check_optipng_installed():
print("\nOptimizing PNG files with optipng...")
for i, file_path in enumerate(temp_files, 1):
print(f"[{i}/{total_files}] Optimizing {os.path.basename(file_path)}...", end='', flush=True)
if run_optipng(file_path):
successful_opts += 1
print(" ")
else:
print(" ")
else:
print("\nNote: optipng not found. Skipping PNG optimization.")
# Print summary
print("\nOptimization complete!")
print(f"Successfully trimmed: {successful_trims}/{total_files} images")
if check_optipng_installed():
print(f"Successfully optimized: {successful_opts}/{total_files} images")
@cli.command()
@click.argument('output_pdf', type=click.Path())
def finalize(output_pdf):
"""Combine processed pages into final PDF and clean up."""
temp_dir = ensure_temp_dir()
temp_files = get_temp_files()
if not temp_files:
print("No pages found in temporary directory. Run 'extract' first.")
return
print(f"Combining {len(temp_files)} pages into {output_pdf}...")
# Convert to PDF
with open(output_pdf, "wb") as f:
f.write(img2pdf.convert(temp_files))
# Clean up temporary files
print("Cleaning up temporary files...")
for file_path in temp_files:
os.remove(file_path)
os.rmdir(temp_dir)
print("PDF created successfully and temporary files removed!")
if __name__ == '__main__':
cli()