refactor: consolidate image trimming and standardize processing
- Replace OpenCV-based trim_whitespace with PIL implementation - Remove redundant trim_whitespace_pil function - Add initial trim step to extract command - Standardize image processing to 2048px wide grayscale using Lanczos
This commit is contained in:
+68
-63
@@ -14,6 +14,7 @@ from pdf2image import convert_from_path
|
|||||||
|
|
||||||
import settings
|
import settings
|
||||||
from settings import OptimizationLevel
|
from settings import OptimizationLevel
|
||||||
|
from PIL import Image
|
||||||
|
|
||||||
class Settings:
|
class Settings:
|
||||||
TRIM_PADDING_PIXELS = 20
|
TRIM_PADDING_PIXELS = 20
|
||||||
@@ -52,12 +53,79 @@ def extract(input_pdf):
|
|||||||
|
|
||||||
# Save each page
|
# Save each page
|
||||||
for i, page in enumerate(pages):
|
for i, page in enumerate(pages):
|
||||||
|
# Convert to grayscale
|
||||||
|
page = page.convert('L')
|
||||||
|
|
||||||
output_path = os.path.join(temp_dir, f"page_{i:03d}.png")
|
output_path = os.path.join(temp_dir, f"page_{i:03d}.png")
|
||||||
|
|
||||||
|
# Save initial version
|
||||||
|
page.save(output_path, "PNG", optimize=False)
|
||||||
|
|
||||||
|
# Trim whitespace
|
||||||
|
trim_whitespace(output_path)
|
||||||
|
|
||||||
|
# Reload the trimmed image
|
||||||
|
page = Image.open(output_path)
|
||||||
|
|
||||||
|
# Calculate new height maintaining aspect ratio
|
||||||
|
width = 2048
|
||||||
|
ratio = width / page.width
|
||||||
|
height = int(page.height * ratio)
|
||||||
|
|
||||||
|
# Resize using Lanczos
|
||||||
|
page = page.resize((width, height), Image.Resampling.LANCZOS)
|
||||||
|
|
||||||
|
# Save final version
|
||||||
page.save(output_path, "PNG", optimize=False)
|
page.save(output_path, "PNG", optimize=False)
|
||||||
print(f"Saved page {i+1}/{len(pages)}")
|
print(f"Saved page {i+1}/{len(pages)}")
|
||||||
|
|
||||||
print(f"Extracted {len(pages)} pages to {temp_dir}/")
|
print(f"Extracted {len(pages)} pages to {temp_dir}/")
|
||||||
|
|
||||||
|
def trim_whitespace(image_path: str) -> bool:
|
||||||
|
"""Remove white space from around the image.
|
||||||
|
|
||||||
|
Handles both RGB and RGBA images, treating transparent pixels as white.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
image_path: Path to the image file
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
bool: True if successful, False otherwise
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
# Open image with PIL
|
||||||
|
image = Image.open(image_path)
|
||||||
|
|
||||||
|
# Convert to grayscale if not already
|
||||||
|
if image.mode != 'L':
|
||||||
|
image = image.convert('L')
|
||||||
|
|
||||||
|
# Get the bounding box of non-white pixels
|
||||||
|
bbox = image.getbbox()
|
||||||
|
if not bbox:
|
||||||
|
print(f"Warning: No content found in {image_path}")
|
||||||
|
return False
|
||||||
|
|
||||||
|
# Add padding
|
||||||
|
padding = settings.TRIM_PADDING_PIXELS
|
||||||
|
width, height = image.size
|
||||||
|
x1, y1, x2, y2 = bbox
|
||||||
|
x1 = max(0, x1 - padding)
|
||||||
|
y1 = max(0, y1 - padding)
|
||||||
|
x2 = min(width, x2 + padding)
|
||||||
|
y2 = min(height, y2 + padding)
|
||||||
|
|
||||||
|
# Crop the image
|
||||||
|
image = image.crop((x1, y1, x2, y2))
|
||||||
|
|
||||||
|
# Save the image
|
||||||
|
image.save(image_path, "PNG", optimize=False)
|
||||||
|
return True
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
print(f"Error processing {image_path}: {str(e)}")
|
||||||
|
return False
|
||||||
|
|
||||||
@cli.command()
|
@cli.command()
|
||||||
def deskew():
|
def deskew():
|
||||||
"""Deskew all pages in temporary directory."""
|
"""Deskew all pages in temporary directory."""
|
||||||
@@ -132,69 +200,6 @@ def deskew():
|
|||||||
else:
|
else:
|
||||||
print(" No line segments detected")
|
print(" No line segments detected")
|
||||||
|
|
||||||
def trim_whitespace(image_path: str) -> bool:
|
|
||||||
"""Remove white space from around the image.
|
|
||||||
|
|
||||||
Handles both RGB and RGBA images, treating transparent pixels as white.
|
|
||||||
|
|
||||||
Returns:
|
|
||||||
bool: True if successful, False otherwise
|
|
||||||
"""
|
|
||||||
try:
|
|
||||||
# Read the image with alpha channel
|
|
||||||
img = cv2.imread(image_path, cv2.IMREAD_UNCHANGED)
|
|
||||||
if img is None:
|
|
||||||
print(f"Warning: Could not read image {image_path}")
|
|
||||||
return False
|
|
||||||
|
|
||||||
# Convert to grayscale, handling transparency
|
|
||||||
if img.shape[-1] == 4: # RGBA
|
|
||||||
# Create a white background
|
|
||||||
white_background = np.ones_like(img, dtype=np.uint8) * 255
|
|
||||||
|
|
||||||
# Extract alpha channel and create mask
|
|
||||||
alpha = img[:, :, 3]
|
|
||||||
alpha_mask = alpha[:, :, np.newaxis] / 255.0
|
|
||||||
|
|
||||||
# Blend image with white background based on alpha
|
|
||||||
img = (img[:, :, :3] * alpha_mask + white_background[:, :, :3] * (1 - alpha_mask)).astype(np.uint8)
|
|
||||||
|
|
||||||
# Convert to grayscale
|
|
||||||
gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)
|
|
||||||
|
|
||||||
# Threshold the image
|
|
||||||
_, thresh = cv2.threshold(gray, 250, 255, cv2.THRESH_BINARY_INV)
|
|
||||||
|
|
||||||
# Find non-zero points
|
|
||||||
coords = cv2.findNonZero(thresh)
|
|
||||||
if coords is None:
|
|
||||||
print(f"Warning: No content found in {image_path}")
|
|
||||||
return False
|
|
||||||
|
|
||||||
# Get bounding rectangle
|
|
||||||
x, y, w, h = cv2.boundingRect(coords)
|
|
||||||
|
|
||||||
# Add padding
|
|
||||||
padding = settings.TRIM_PADDING_PIXELS
|
|
||||||
height, width = img.shape[:2]
|
|
||||||
x = max(0, x - padding)
|
|
||||||
y = max(0, y - padding)
|
|
||||||
w = min(width - x, w + 2 * padding)
|
|
||||||
h = min(height - y, h + 2 * padding)
|
|
||||||
|
|
||||||
# Crop the image
|
|
||||||
if img.shape[-1] == 4: # If original was RGBA
|
|
||||||
cropped = img[:, :, :4][y:y+h, x:x+w] # Keep alpha channel
|
|
||||||
else:
|
|
||||||
cropped = img[y:y+h, x:x+w]
|
|
||||||
|
|
||||||
# Save the cropped image with transparency preserved
|
|
||||||
cv2.imwrite(image_path, cropped)
|
|
||||||
return True
|
|
||||||
except Exception as e:
|
|
||||||
print(f"Error processing {image_path}: {str(e)}")
|
|
||||||
return False
|
|
||||||
|
|
||||||
def convert_to_monochrome(image_path: str) -> bool:
|
def convert_to_monochrome(image_path: str) -> bool:
|
||||||
"""Convert image to 1-bit monochrome.
|
"""Convert image to 1-bit monochrome.
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user