Merge branch 'feature/music-score-cleanup-v2'
This commit is contained in:
@@ -10,3 +10,4 @@ wheels/
|
|||||||
.venv
|
.venv
|
||||||
|
|
||||||
*.pdf
|
*.pdf
|
||||||
|
*.png
|
||||||
@@ -0,0 +1,103 @@
|
|||||||
|
# PDF Musical Score Cleaner
|
||||||
|
|
||||||
|
A command-line tool for processing and cleaning scanned musical score PDFs. This tool helps you extract, deskew, optimize, and recompile PDF files while maintaining high quality and readability of musical notation.
|
||||||
|
|
||||||
|
## Features
|
||||||
|
|
||||||
|
- **Page Extraction**: Extract individual pages from PDF files
|
||||||
|
- **Deskewing**: Automatically correct page rotation using staff line detection
|
||||||
|
- **White Space Trimming**: Remove excess white space around the musical content
|
||||||
|
- **PNG Optimization**: Optimize PNG files using optipng (if installed)
|
||||||
|
- **Modular Processing**: Process your files step by step or all at once
|
||||||
|
- **High Quality Output**: Preserve image quality throughout the process
|
||||||
|
|
||||||
|
## Installation
|
||||||
|
|
||||||
|
1. Ensure you have Python 3.8+ installed
|
||||||
|
2. Install uv (recommended) or pip
|
||||||
|
3. Clone this repository:
|
||||||
|
```bash
|
||||||
|
git clone <repository-url>
|
||||||
|
cd notes_cleaner
|
||||||
|
```
|
||||||
|
4. Install dependencies:
|
||||||
|
```bash
|
||||||
|
uv sync
|
||||||
|
```
|
||||||
|
|
||||||
|
5. (Optional) Install optipng for additional PNG optimization:
|
||||||
|
```bash
|
||||||
|
# Ubuntu/Debian
|
||||||
|
sudo apt-get install optipng
|
||||||
|
|
||||||
|
# macOS
|
||||||
|
brew install optipng
|
||||||
|
|
||||||
|
# Arch Linux
|
||||||
|
sudo pacman -Sy optipng
|
||||||
|
```
|
||||||
|
|
||||||
|
## Usage
|
||||||
|
|
||||||
|
The tool provides several commands that can be run independently:
|
||||||
|
|
||||||
|
### Extract Pages
|
||||||
|
```bash
|
||||||
|
./pdf_cleaner.py extract input.pdf
|
||||||
|
```
|
||||||
|
Extracts all pages from the input PDF to a temporary directory.
|
||||||
|
|
||||||
|
### Deskew Pages
|
||||||
|
```bash
|
||||||
|
./pdf_cleaner.py deskew
|
||||||
|
```
|
||||||
|
Automatically detects and corrects page rotation by analyzing staff lines.
|
||||||
|
|
||||||
|
### Optimize Pages
|
||||||
|
```bash
|
||||||
|
./pdf_cleaner.py optimize
|
||||||
|
```
|
||||||
|
Trims excess white space and optionally runs PNG optimization (requires optipng).
|
||||||
|
|
||||||
|
### Create Final PDF
|
||||||
|
```bash
|
||||||
|
./pdf_cleaner.py finalize output.pdf
|
||||||
|
```
|
||||||
|
Combines all processed pages into a final PDF and cleans up temporary files.
|
||||||
|
|
||||||
|
### Typical Workflow
|
||||||
|
```bash
|
||||||
|
./pdf_cleaner.py extract input.pdf # Extract pages
|
||||||
|
./pdf_cleaner.py deskew # Correct rotation
|
||||||
|
./pdf_cleaner.py optimize # Remove white space and optimize
|
||||||
|
./pdf_cleaner.py finalize output.pdf # Create final PDF
|
||||||
|
```
|
||||||
|
|
||||||
|
## How It Works
|
||||||
|
|
||||||
|
1. **Extraction**: Uses pdf2image to convert PDF pages to high-quality PNG images
|
||||||
|
2. **Deskewing**:
|
||||||
|
- Applies morphological operations to enhance horizontal lines
|
||||||
|
- Uses Hough transform to detect staff lines
|
||||||
|
- Calculates and corrects rotation based on detected lines
|
||||||
|
3. **Optimization**:
|
||||||
|
- Detects content boundaries and removes excess white space
|
||||||
|
- Optionally runs optipng for additional file size reduction
|
||||||
|
4. **Finalization**: Combines processed images back into a PDF using img2pdf
|
||||||
|
|
||||||
|
## Dependencies
|
||||||
|
|
||||||
|
- click: Command line interface
|
||||||
|
- opencv-python: Image processing and deskewing
|
||||||
|
- numpy: Numerical operations
|
||||||
|
- pdf2image: PDF to image conversion
|
||||||
|
- img2pdf: Image to PDF conversion
|
||||||
|
- optipng (optional): PNG file optimization
|
||||||
|
|
||||||
|
## Contributing
|
||||||
|
|
||||||
|
Contributions are welcome! Please feel free to submit a Pull Request.
|
||||||
|
|
||||||
|
## License
|
||||||
|
|
||||||
|
[Insert chosen license here]
|
||||||
Regular → Executable
+369
-118
@@ -1,157 +1,408 @@
|
|||||||
#!/usr/bin/env python3
|
#!/usr/bin/env -S uv run
|
||||||
|
|
||||||
import os
|
import os
|
||||||
|
import shutil
|
||||||
|
import subprocess
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import List, Tuple
|
||||||
|
|
||||||
|
import click
|
||||||
import cv2
|
import cv2
|
||||||
|
import img2pdf
|
||||||
import numpy as np
|
import numpy as np
|
||||||
from pdf2image import convert_from_path
|
from pdf2image import convert_from_path
|
||||||
import img2pdf
|
|
||||||
|
import settings
|
||||||
|
from settings import OptimizationLevel
|
||||||
from PIL import Image
|
from PIL import Image
|
||||||
import argparse
|
|
||||||
|
|
||||||
def deskew(image):
|
class Settings:
|
||||||
"""Deskew the image using contour detection and rotation."""
|
TRIM_PADDING_PIXELS = 20
|
||||||
# Create a copy for processing while keeping original quality
|
MONOCHROME_THRESHOLD = 127
|
||||||
proc_image = image.copy()
|
OPTIPNG_OPTIMIZATION_LEVEL = 7
|
||||||
|
PDF_BORDER_SIZE = 50
|
||||||
|
|
||||||
# Convert to grayscale and blur
|
settings = Settings()
|
||||||
gray = cv2.cvtColor(proc_image, cv2.COLOR_BGR2GRAY)
|
|
||||||
blur = cv2.GaussianBlur(gray, (9, 9), 0)
|
|
||||||
|
|
||||||
# Threshold the image
|
def ensure_temp_dir():
|
||||||
thresh = cv2.threshold(blur, 0, 255, cv2.THRESH_BINARY_INV + cv2.THRESH_OTSU)[1]
|
"""Ensure temporary directory exists and return its path."""
|
||||||
|
temp_dir = "temp_processed_images"
|
||||||
|
os.makedirs(temp_dir, exist_ok=True)
|
||||||
|
return temp_dir
|
||||||
|
|
||||||
# Find all contours
|
def get_temp_files():
|
||||||
contours, _ = cv2.findContours(thresh, cv2.RETR_LIST, cv2.CHAIN_APPROX_SIMPLE)
|
"""Get list of temporary PNG files in order."""
|
||||||
|
temp_dir = ensure_temp_dir()
|
||||||
|
files = [f for f in os.listdir(temp_dir) if f.endswith('.png')]
|
||||||
|
files.sort() # Ensure correct page order
|
||||||
|
return [os.path.join(temp_dir, f) for f in files]
|
||||||
|
|
||||||
if not contours:
|
@click.group()
|
||||||
return image
|
def cli():
|
||||||
|
"""PDF cleaning toolbox for musical scores."""
|
||||||
|
pass
|
||||||
|
|
||||||
# Find largest contour
|
@cli.command()
|
||||||
contour = max(contours, key=cv2.contourArea)
|
@click.argument('input_pdf', type=click.Path(exists=True))
|
||||||
|
def extract(input_pdf):
|
||||||
|
"""Extract pages from PDF to temporary directory."""
|
||||||
|
temp_dir = ensure_temp_dir()
|
||||||
|
|
||||||
# Find minimum area rectangle
|
# Convert PDF to images
|
||||||
rect = cv2.minAreaRect(contour)
|
print(f"Extracting pages from {input_pdf}...")
|
||||||
angle = rect[-1]
|
pages = convert_from_path(input_pdf, dpi=400)
|
||||||
|
|
||||||
# Adjust angle to be between -45 and 45 degrees
|
# Save each page
|
||||||
while angle < -45:
|
for i, page in enumerate(pages):
|
||||||
angle += 90
|
# Convert to grayscale
|
||||||
while angle > 45:
|
page = page.convert('L')
|
||||||
angle -= 90
|
|
||||||
|
|
||||||
# Only rotate if the angle is significant enough
|
output_path = os.path.join(temp_dir, f"page_{i:03d}.png")
|
||||||
if abs(angle) < 0.5: # Skip tiny rotations
|
|
||||||
return image
|
|
||||||
|
|
||||||
# Rotate the image
|
# Save initial version
|
||||||
(h, w) = image.shape[:2]
|
page.save(output_path, "PNG", optimize=False)
|
||||||
center = (w // 2, h // 2)
|
|
||||||
M = cv2.getRotationMatrix2D(center, angle, 1.0)
|
# Trim whitespace
|
||||||
rotated = cv2.warpAffine(image, M, (w, h),
|
trim_whitespace(output_path)
|
||||||
|
|
||||||
|
# Reload the trimmed image
|
||||||
|
page = Image.open(output_path)
|
||||||
|
|
||||||
|
# Calculate new height maintaining aspect ratio
|
||||||
|
width = 2048
|
||||||
|
ratio = width / page.width
|
||||||
|
height = int(page.height * ratio)
|
||||||
|
|
||||||
|
# Resize using Lanczos
|
||||||
|
page = page.resize((width, height), Image.Resampling.LANCZOS)
|
||||||
|
|
||||||
|
# Save final version
|
||||||
|
page.save(output_path, "PNG", optimize=False)
|
||||||
|
print(f"Saved page {i+1}/{len(pages)}")
|
||||||
|
|
||||||
|
print(f"Extracted {len(pages)} pages to {temp_dir}/")
|
||||||
|
|
||||||
|
def trim_whitespace(image_path: str, is_final: bool = False) -> bool:
|
||||||
|
"""Remove white space from around the image.
|
||||||
|
|
||||||
|
Handles both RGB and RGBA images, treating transparent pixels as white.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
image_path: Path to the image file
|
||||||
|
is_final: Whether this is the final operation on the image
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
bool: True if successful, False otherwise
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
# Open image with PIL
|
||||||
|
image = Image.open(image_path)
|
||||||
|
|
||||||
|
# If image has transparency, flatten it first
|
||||||
|
if image.mode == 'RGBA':
|
||||||
|
# Create a white background
|
||||||
|
background = Image.new('RGB', image.size, 'white')
|
||||||
|
# Paste using alpha channel as mask
|
||||||
|
background.paste(image, mask=image.split()[3])
|
||||||
|
image = background
|
||||||
|
|
||||||
|
# Convert to grayscale
|
||||||
|
image = image.convert('L')
|
||||||
|
|
||||||
|
# Threshold to make all light pixels white and everything else black
|
||||||
|
# This helps with finding content bounds
|
||||||
|
image = image.point(lambda x: 255 if x > 250 else 0)
|
||||||
|
|
||||||
|
# Invert so content is white on black background
|
||||||
|
image = Image.eval(image, lambda x: 255 - x)
|
||||||
|
|
||||||
|
# Get the bounding box of content (now white pixels)
|
||||||
|
bbox = image.getbbox()
|
||||||
|
if not bbox:
|
||||||
|
print(f"Warning: No content found in {image_path}")
|
||||||
|
return False
|
||||||
|
|
||||||
|
# Add padding
|
||||||
|
padding = settings.TRIM_PADDING_PIXELS
|
||||||
|
width, height = image.size
|
||||||
|
x1, y1, x2, y2 = bbox
|
||||||
|
x1 = max(0, x1 - padding)
|
||||||
|
y1 = max(0, y1 - padding)
|
||||||
|
x2 = min(width, x2 + padding)
|
||||||
|
y2 = min(height, y2 + padding)
|
||||||
|
|
||||||
|
# Open original image again and crop it using the bounds
|
||||||
|
original = Image.open(image_path)
|
||||||
|
cropped = original.crop((x1, y1, x2, y2))
|
||||||
|
|
||||||
|
# Save the cropped image, optimizing only if this is the final operation
|
||||||
|
cropped.save(image_path, "PNG", optimize=is_final)
|
||||||
|
return True
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
print(f"Error processing {image_path}: {str(e)}")
|
||||||
|
return False
|
||||||
|
|
||||||
|
@cli.command()
|
||||||
|
def deskew():
|
||||||
|
"""Deskew all pages in temporary directory."""
|
||||||
|
temp_files = get_temp_files()
|
||||||
|
if not temp_files:
|
||||||
|
print("No pages found in temporary directory. Run 'extract' first.")
|
||||||
|
return
|
||||||
|
|
||||||
|
for file_path in temp_files:
|
||||||
|
print(f"Deskewing {os.path.basename(file_path)}...")
|
||||||
|
|
||||||
|
# Read image
|
||||||
|
image = cv2.imread(file_path)
|
||||||
|
|
||||||
|
# Convert to grayscale
|
||||||
|
gray = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)
|
||||||
|
|
||||||
|
# Apply threshold to get binary image
|
||||||
|
_, binary = cv2.threshold(gray, 127, 255, cv2.THRESH_BINARY)
|
||||||
|
|
||||||
|
# Create a rectangular kernel that's wider than it is tall
|
||||||
|
# This helps detect horizontal lines
|
||||||
|
kernel_length = np.array(binary).shape[1]//80
|
||||||
|
horizontal_kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (kernel_length, 1))
|
||||||
|
|
||||||
|
# Detect horizontal lines
|
||||||
|
horizontal_lines = cv2.erode(binary, horizontal_kernel, iterations=3)
|
||||||
|
horizontal_lines = cv2.dilate(horizontal_lines, horizontal_kernel, iterations=3)
|
||||||
|
|
||||||
|
# Use probabilistic Hough transform to detect line segments
|
||||||
|
line_segments = cv2.HoughLinesP(
|
||||||
|
cv2.bitwise_not(horizontal_lines),
|
||||||
|
rho=1,
|
||||||
|
theta=np.pi/180,
|
||||||
|
threshold=100,
|
||||||
|
minLineLength=binary.shape[1]//4, # Lines must be at least 1/4 of image width
|
||||||
|
maxLineGap=20
|
||||||
|
)
|
||||||
|
|
||||||
|
if line_segments is not None and len(line_segments) > 0:
|
||||||
|
# Calculate angles of detected line segments
|
||||||
|
angles = []
|
||||||
|
for line in line_segments:
|
||||||
|
x1, y1, x2, y2 = line[0]
|
||||||
|
if x2 - x1 == 0: # Avoid division by zero
|
||||||
|
continue
|
||||||
|
angle = np.degrees(np.arctan2(y2 - y1, x2 - x1))
|
||||||
|
# Only consider angles that are close to horizontal
|
||||||
|
if abs(angle) < 20:
|
||||||
|
angles.append(angle)
|
||||||
|
|
||||||
|
if angles:
|
||||||
|
# Use median angle to avoid outliers
|
||||||
|
median_angle = np.median(angles)
|
||||||
|
|
||||||
|
# Only rotate if the angle is significant but not too large
|
||||||
|
if 0.5 < abs(median_angle) < 20:
|
||||||
|
height, width = image.shape[:2]
|
||||||
|
center = (width/2, height/2)
|
||||||
|
rotation_matrix = cv2.getRotationMatrix2D(center, median_angle, 1.0)
|
||||||
|
rotated = cv2.warpAffine(image, rotation_matrix, (width, height),
|
||||||
flags=cv2.INTER_CUBIC,
|
flags=cv2.INTER_CUBIC,
|
||||||
borderMode=cv2.BORDER_REPLICATE)
|
borderMode=cv2.BORDER_REPLICATE)
|
||||||
|
|
||||||
return rotated
|
# Save rotated image
|
||||||
|
cv2.imwrite(file_path, rotated)
|
||||||
def adjust_levels(image):
|
print(f" Rotated by {median_angle:.2f} degrees")
|
||||||
"""Adjust image levels for better contrast and clarity."""
|
|
||||||
# Convert to grayscale if not already
|
|
||||||
if len(image.shape) == 3:
|
|
||||||
gray = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)
|
|
||||||
else:
|
else:
|
||||||
gray = image
|
print(" No significant rotation needed")
|
||||||
|
else:
|
||||||
|
print(" No valid horizontal lines found")
|
||||||
|
else:
|
||||||
|
print(" No line segments detected")
|
||||||
|
|
||||||
# Apply bilateral filter to preserve edges while reducing noise
|
def convert_to_monochrome(image_path: str, is_final: bool = False) -> bool:
|
||||||
denoised = cv2.bilateralFilter(gray, 9, 75, 75)
|
"""Convert image to 1-bit monochrome.
|
||||||
|
|
||||||
# Create a background mask using morphological operations
|
Handles RGBA images by converting transparent pixels to white before thresholding.
|
||||||
kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (15, 15))
|
|
||||||
background = cv2.morphologyEx(denoised, cv2.MORPH_DILATE, kernel)
|
|
||||||
|
|
||||||
# Subtract background to normalize lighting
|
Args:
|
||||||
normalized = cv2.subtract(background, denoised)
|
image_path: Path to the image file
|
||||||
|
is_final: Whether this is the final operation on the image
|
||||||
|
|
||||||
# Apply Gaussian blur to reduce noise while preserving edges
|
Returns:
|
||||||
blurred = cv2.GaussianBlur(normalized, (3, 3), 0)
|
bool: True if successful, False otherwise
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
# Open image with PIL
|
||||||
|
image = Image.open(image_path)
|
||||||
|
|
||||||
# Use Otsu's thresholding for optimal binary threshold
|
# Convert to RGBA if not already
|
||||||
_, binary = cv2.threshold(blurred, 0, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU)
|
if image.mode != 'RGBA':
|
||||||
|
image = image.convert('RGBA')
|
||||||
|
|
||||||
# Always ensure background is white and text is black
|
# Get the image data as a list of pixels
|
||||||
if cv2.countNonZero(binary) < binary.size / 2:
|
data = image.getdata()
|
||||||
binary = cv2.bitwise_not(binary)
|
|
||||||
|
|
||||||
return binary
|
# Create new image data, replacing transparent pixels with white
|
||||||
|
new_data = []
|
||||||
|
for item in data:
|
||||||
|
# If pixel is transparent (alpha < 128), make it white
|
||||||
|
if item[3] < 128:
|
||||||
|
new_data.append((255, 255, 255, 255))
|
||||||
|
else:
|
||||||
|
new_data.append(item)
|
||||||
|
|
||||||
def process_pdf(input_path, output_path):
|
# Create new image with modified data
|
||||||
"""Process a PDF file and save the cleaned version."""
|
image.putdata(new_data)
|
||||||
print(f"Processing {input_path}...")
|
|
||||||
|
|
||||||
# Convert PDF to images with high DPI to ensure minimum width of 2048px
|
# Convert to grayscale
|
||||||
pages = convert_from_path(input_path, dpi=300)
|
image = image.convert('L')
|
||||||
|
|
||||||
# Create temporary directory for processed images
|
# Convert to 1-bit using threshold
|
||||||
temp_dir = "temp_processed_images"
|
image = image.point(lambda x: 255 if x > settings.MONOCHROME_THRESHOLD else 0, '1')
|
||||||
os.makedirs(temp_dir, exist_ok=True)
|
|
||||||
|
|
||||||
# Process each page
|
# Save the monochrome image, optimizing only if this is the final operation
|
||||||
temp_image_paths = []
|
image.save(image_path, "PNG", optimize=is_final)
|
||||||
for i, page in enumerate(pages):
|
return True
|
||||||
print(f"Processing page {i+1}/{len(pages)}")
|
|
||||||
|
|
||||||
# Ensure minimum width of 2048px
|
except Exception as e:
|
||||||
width, height = page.size
|
print(f"Error converting to monochrome {image_path}: {str(e)}")
|
||||||
scale = max(1, 2048 / width)
|
return False
|
||||||
if scale > 1:
|
|
||||||
new_width = int(width * scale)
|
|
||||||
new_height = int(height * scale)
|
|
||||||
page = page.resize((new_width, new_height), Image.Resampling.LANCZOS)
|
|
||||||
|
|
||||||
# Convert PIL Image to OpenCV format
|
def check_optipng_installed() -> bool:
|
||||||
opencv_image = cv2.cvtColor(np.array(page), cv2.COLOR_RGB2BGR)
|
"""Check if optipng is installed."""
|
||||||
|
try:
|
||||||
|
result = subprocess.run(['optipng', '-v'], capture_output=True, text=True)
|
||||||
|
return result.returncode == 0
|
||||||
|
except FileNotFoundError:
|
||||||
|
return False
|
||||||
|
|
||||||
# Process the image
|
def run_optipng(file_path: str) -> bool:
|
||||||
deskewed = deskew(opencv_image)
|
"""Run optipng on a file with error handling.
|
||||||
cleaned = adjust_levels(deskewed)
|
|
||||||
|
|
||||||
# Convert to PIL Image
|
Returns:
|
||||||
pil_image = Image.fromarray(cleaned)
|
bool: True if successful, False otherwise
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
result = subprocess.run(
|
||||||
|
['optipng', f'-o{settings.OPTIPNG_OPTIMIZATION_LEVEL}', file_path],
|
||||||
|
capture_output=True,
|
||||||
|
text=True,
|
||||||
|
check=True
|
||||||
|
)
|
||||||
|
return True
|
||||||
|
except subprocess.CalledProcessError as e:
|
||||||
|
print(f"Error optimizing {file_path}: {e.stderr}")
|
||||||
|
return False
|
||||||
|
except Exception as e:
|
||||||
|
print(f"Unexpected error optimizing {file_path}: {str(e)}")
|
||||||
|
return False
|
||||||
|
|
||||||
# Save temporarily as 1-bit PNG
|
@cli.command()
|
||||||
temp_path = os.path.join(temp_dir, f"page_{i:03d}.png")
|
@click.option('--level', type=click.IntRange(1, 3), default=1,
|
||||||
pil_image.save(temp_path, "PNG", optimize=False)
|
help='Optimization level: 1=trim, 2=monochrome, 3=full with optipng')
|
||||||
temp_image_paths.append(temp_path)
|
def optimize(level):
|
||||||
|
"""Optimize images with specified level of processing.
|
||||||
|
|
||||||
# Save processed images as PDF with high quality settings
|
Optimization Levels:
|
||||||
print("Saving cleaned PDF...")
|
1: Only trim whitespace
|
||||||
a4_width_mm = 210
|
2: Level 1 + convert to 1-bit monochrome
|
||||||
a4_height_mm = 297
|
3: Level 2 + optipng optimization
|
||||||
layout_fun = img2pdf.get_layout_fun((img2pdf.mm_to_pt(a4_width_mm),
|
"""
|
||||||
img2pdf.mm_to_pt(a4_height_mm)))
|
temp_dir = ensure_temp_dir()
|
||||||
|
temp_files = get_temp_files()
|
||||||
with open(output_path, "wb") as f:
|
if not temp_files:
|
||||||
f.write(img2pdf.convert(temp_image_paths,
|
print("No pages found in temporary directory. Run 'extract' first.")
|
||||||
layout_fun=layout_fun,
|
|
||||||
with_pdfrw=True))
|
|
||||||
|
|
||||||
# Clean up temporary files
|
|
||||||
for temp_path in temp_image_paths:
|
|
||||||
os.remove(temp_path)
|
|
||||||
os.rmdir(temp_dir)
|
|
||||||
|
|
||||||
print(f"Saved cleaned PDF to {output_path}")
|
|
||||||
|
|
||||||
def main():
|
|
||||||
parser = argparse.ArgumentParser(description='Clean and straighten image-based PDFs')
|
|
||||||
parser.add_argument('input_pdf', help='Path to the input PDF file')
|
|
||||||
parser.add_argument('output_pdf', help='Path for the output PDF file')
|
|
||||||
|
|
||||||
args = parser.parse_args()
|
|
||||||
|
|
||||||
if not os.path.exists(args.input_pdf):
|
|
||||||
print(f"Error: Input file '{args.input_pdf}' does not exist")
|
|
||||||
return
|
return
|
||||||
|
|
||||||
process_pdf(args.input_pdf, args.output_pdf)
|
opt_level = OptimizationLevel(level)
|
||||||
|
|
||||||
if __name__ == "__main__":
|
total_files = len(temp_files)
|
||||||
main()
|
successful = {
|
||||||
|
'trim': 0,
|
||||||
|
'monochrome': 0,
|
||||||
|
'optipng': 0
|
||||||
|
}
|
||||||
|
|
||||||
|
print(f"Processing {total_files} images at optimization level {level}...")
|
||||||
|
|
||||||
|
# Step 1: Always trim whitespace
|
||||||
|
print("\nTrimming whitespace from images...")
|
||||||
|
for i, file_path in enumerate(temp_files, 1):
|
||||||
|
print(f"[{i}/{total_files}] Processing {os.path.basename(file_path)}...", end='', flush=True)
|
||||||
|
# Only optimize if this is the final step (level 1)
|
||||||
|
if trim_whitespace(file_path, is_final=(opt_level == OptimizationLevel.TRIM)):
|
||||||
|
successful['trim'] += 1
|
||||||
|
print(" ")
|
||||||
|
else:
|
||||||
|
print(" ")
|
||||||
|
|
||||||
|
# Step 2: Convert to monochrome if level >= 2
|
||||||
|
if int(opt_level) >= int(OptimizationLevel.MONOCHROME):
|
||||||
|
print("\nConverting to monochrome...")
|
||||||
|
for i, file_path in enumerate(temp_files, 1):
|
||||||
|
print(f"[{i}/{total_files}] Converting {os.path.basename(file_path)}...", end='', flush=True)
|
||||||
|
# Only optimize if this is the final step (level 2)
|
||||||
|
if convert_to_monochrome(file_path, is_final=(opt_level == OptimizationLevel.MONOCHROME)):
|
||||||
|
successful['monochrome'] += 1
|
||||||
|
print(" ")
|
||||||
|
else:
|
||||||
|
print(" ")
|
||||||
|
|
||||||
|
# Step 3: Run optipng if level = 3
|
||||||
|
if opt_level == OptimizationLevel.FULL:
|
||||||
|
if check_optipng_installed():
|
||||||
|
print("\nOptimizing PNG files with optipng...")
|
||||||
|
for i, file_path in enumerate(temp_files, 1):
|
||||||
|
print(f"[{i}/{total_files}] Optimizing {os.path.basename(file_path)}...", end='', flush=True)
|
||||||
|
if run_optipng(file_path):
|
||||||
|
successful['optipng'] += 1
|
||||||
|
print(" ")
|
||||||
|
else:
|
||||||
|
print(" ")
|
||||||
|
else:
|
||||||
|
print("\nNote: optipng not found. Skipping PNG optimization.")
|
||||||
|
|
||||||
|
# Print summary
|
||||||
|
print("\nOptimization complete!")
|
||||||
|
print(f"Successfully trimmed: {successful['trim']}/{total_files} images")
|
||||||
|
if int(opt_level) >= int(OptimizationLevel.MONOCHROME):
|
||||||
|
print(f"Successfully converted to monochrome: {successful['monochrome']}/{total_files} images")
|
||||||
|
if opt_level == OptimizationLevel.FULL and check_optipng_installed():
|
||||||
|
print(f"Successfully optimized with optipng: {successful['optipng']}/{total_files} images")
|
||||||
|
|
||||||
|
@cli.command()
|
||||||
|
@click.argument('output_pdf', type=click.Path())
|
||||||
|
def finalize(output_pdf):
|
||||||
|
"""Combine processed pages into final PDF and clean up."""
|
||||||
|
temp_dir = ensure_temp_dir()
|
||||||
|
temp_files = get_temp_files()
|
||||||
|
if not temp_files:
|
||||||
|
print("No pages found in temporary directory. Run 'extract' first.")
|
||||||
|
return
|
||||||
|
|
||||||
|
print(f"Combining {len(temp_files)} pages into {output_pdf}...")
|
||||||
|
|
||||||
|
# A4 size in millimeters
|
||||||
|
A4_WIDTH_MM = 210
|
||||||
|
A4_HEIGHT_MM = 297
|
||||||
|
|
||||||
|
# Convert to PDF with border
|
||||||
|
with open(output_pdf, "wb") as f:
|
||||||
|
f.write(img2pdf.convert(
|
||||||
|
temp_files,
|
||||||
|
with_pdfrw=True,
|
||||||
|
layout_fun=img2pdf.get_layout_fun(
|
||||||
|
pagesize=(img2pdf.mm_to_pt(A4_WIDTH_MM), img2pdf.mm_to_pt(A4_HEIGHT_MM)),
|
||||||
|
border=(settings.PDF_BORDER_SIZE,) * 4, # Same border size for all sides (top, right, bottom, left)
|
||||||
|
fit=img2pdf.FitMode.into
|
||||||
|
)
|
||||||
|
))
|
||||||
|
|
||||||
|
# Clean up temporary files
|
||||||
|
print("Cleaning up temporary files...")
|
||||||
|
for file_path in temp_files:
|
||||||
|
os.remove(file_path)
|
||||||
|
os.rmdir(temp_dir)
|
||||||
|
|
||||||
|
print("PDF created successfully and temporary files removed!")
|
||||||
|
|
||||||
|
if __name__ == '__main__':
|
||||||
|
cli()
|
||||||
|
|||||||
@@ -5,5 +5,6 @@ description = "Add your description here"
|
|||||||
readme = "README.md"
|
readme = "README.md"
|
||||||
requires-python = ">=3.12"
|
requires-python = ">=3.12"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
|
"click>=8.1.7",
|
||||||
"pydantic>=2.10.3",
|
"pydantic>=2.10.3",
|
||||||
]
|
]
|
||||||
|
|||||||
+37
@@ -0,0 +1,37 @@
|
|||||||
|
"""Configuration settings for the PDF Musical Score Cleaner."""
|
||||||
|
|
||||||
|
import os
|
||||||
|
import multiprocessing
|
||||||
|
from enum import IntEnum
|
||||||
|
|
||||||
|
class OptimizationLevel(IntEnum):
|
||||||
|
"""Optimization levels for image processing.
|
||||||
|
|
||||||
|
Levels:
|
||||||
|
1: Only trim whitespace
|
||||||
|
2: Trim + convert to 1-bit monochrome
|
||||||
|
3: All optimizations + optipng
|
||||||
|
"""
|
||||||
|
TRIM = 1 # Only trim whitespace
|
||||||
|
MONOCHROME = 2 # Trim + convert to 1-bit monochrome
|
||||||
|
FULL = 3 # All optimizations + optipng
|
||||||
|
|
||||||
|
# Number of parallel processes to use for operations that support parallelization
|
||||||
|
# Defaults to number of CPU cores - 1, but never less than 1
|
||||||
|
DEFAULT_PARALLEL_PROCESSES = max(1, multiprocessing.cpu_count() - 1)
|
||||||
|
|
||||||
|
# Can be overridden by environment variable
|
||||||
|
PARALLEL_PROCESSES = int(os.getenv('NOTES_CLEANER_PARALLEL_PROCESSES', DEFAULT_PARALLEL_PROCESSES))
|
||||||
|
|
||||||
|
# Border size in pixels for the final PDF output
|
||||||
|
PDF_BORDER_SIZE = 50
|
||||||
|
|
||||||
|
# Optimization settings
|
||||||
|
OPTIPNG_OPTIMIZATION_LEVEL = int(os.getenv('NOTES_CLEANER_OPTIPNG_LEVEL', '7')) # 0-7, higher = better compression but slower
|
||||||
|
TRIM_PADDING_PIXELS = int(os.getenv('NOTES_CLEANER_TRIM_PADDING', '20')) # Padding around content after trimming
|
||||||
|
|
||||||
|
# Image processing settings
|
||||||
|
MONOCHROME_THRESHOLD = int(os.getenv('NOTES_CLEANER_MONO_THRESHOLD', '200')) # 0-255, higher = more white
|
||||||
|
|
||||||
|
# Temporary directory settings
|
||||||
|
TEMP_DIR_PREFIX = 'notes_cleaner_'
|
||||||
@@ -10,16 +10,41 @@ wheels = [
|
|||||||
{ url = "https://files.pythonhosted.org/packages/78/b6/6307fbef88d9b5ee7421e68d78a9f162e0da4900bc5f5793f6d3d0e34fb8/annotated_types-0.7.0-py3-none-any.whl", hash = "sha256:1f02e8b43a8fbbc3f3e0d4f0f4bfc8131bcb4eebe8849b8e5c773f3a1c582a53", size = 13643 },
|
{ url = "https://files.pythonhosted.org/packages/78/b6/6307fbef88d9b5ee7421e68d78a9f162e0da4900bc5f5793f6d3d0e34fb8/annotated_types-0.7.0-py3-none-any.whl", hash = "sha256:1f02e8b43a8fbbc3f3e0d4f0f4bfc8131bcb4eebe8849b8e5c773f3a1c582a53", size = 13643 },
|
||||||
]
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "click"
|
||||||
|
version = "8.1.7"
|
||||||
|
source = { registry = "https://pypi.org/simple" }
|
||||||
|
dependencies = [
|
||||||
|
{ name = "colorama", marker = "platform_system == 'Windows'" },
|
||||||
|
]
|
||||||
|
sdist = { url = "https://files.pythonhosted.org/packages/96/d3/f04c7bfcf5c1862a2a5b845c6b2b360488cf47af55dfa79c98f6a6bf98b5/click-8.1.7.tar.gz", hash = "sha256:ca9853ad459e787e2192211578cc907e7594e294c7ccc834310722b41b9ca6de", size = 336121 }
|
||||||
|
wheels = [
|
||||||
|
{ url = "https://files.pythonhosted.org/packages/00/2e/d53fa4befbf2cfa713304affc7ca780ce4fc1fd8710527771b58311a3229/click-8.1.7-py3-none-any.whl", hash = "sha256:ae74fb96c20a0277a1d615f1e4d73c8414f5a98db8b799a7931d1582f3390c28", size = 97941 },
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "colorama"
|
||||||
|
version = "0.4.6"
|
||||||
|
source = { registry = "https://pypi.org/simple" }
|
||||||
|
sdist = { url = "https://files.pythonhosted.org/packages/d8/53/6f443c9a4a8358a93a6792e2acffb9d9d5cb0a5cfd8802644b7b1c9a02e4/colorama-0.4.6.tar.gz", hash = "sha256:08695f5cb7ed6e0531a20572697297273c47b8cae5a63ffc6d6ed5c201be6e44", size = 27697 }
|
||||||
|
wheels = [
|
||||||
|
{ url = "https://files.pythonhosted.org/packages/d1/d6/3965ed04c63042e047cb6a3e6ed1a63a35087b6a609aa3a15ed8ac56c221/colorama-0.4.6-py2.py3-none-any.whl", hash = "sha256:4f1d9991f5acc0ca119f9d443620b77f9d6b33703e51011c16baf57afb285fc6", size = 25335 },
|
||||||
|
]
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "notes-cleaner"
|
name = "notes-cleaner"
|
||||||
version = "0.1.0"
|
version = "0.1.0"
|
||||||
source = { virtual = "." }
|
source = { virtual = "." }
|
||||||
dependencies = [
|
dependencies = [
|
||||||
|
{ name = "click" },
|
||||||
{ name = "pydantic" },
|
{ name = "pydantic" },
|
||||||
]
|
]
|
||||||
|
|
||||||
[package.metadata]
|
[package.metadata]
|
||||||
requires-dist = [{ name = "pydantic", specifier = ">=2.10.3" }]
|
requires-dist = [
|
||||||
|
{ name = "click", specifier = ">=8.1.7" },
|
||||||
|
{ name = "pydantic", specifier = ">=2.10.3" },
|
||||||
|
]
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "pydantic"
|
name = "pydantic"
|
||||||
|
|||||||
Reference in New Issue
Block a user