Re-engraving is a rescue path for the handful of systems a scan cannot deliver, so the window is an editing surface rather than an automation project. Three full-width rows - the scanned system, the render, the form - because a system is wide and short and the job is comparing one against the other bar by bar. The render is shown scaled to the scan's staff height, which is what export does anyway, so it previews the real thing. A form rather than a text box. Key and time are slice-level, clef, notes and lyrics per voice: every staff in a system carries the same key signature, and Kaipaava proves it across five-staff and two-staff systems alike. Notes and lyrics stay raw LilyPond, so slurs, dynamics, tuplets and the laissezVibrer/repeatTie idiom for ties crossing into the next slice all work untouched. Notes are entered in \relative mode, referenced to the middle of each clef's staff, so a part needs no octave marks at all in the common case. The time signature is used for spacing and bar checks but not printed: the printed score repeats the key at every system and the time only at the first, so a re-engraved middle slice showing one would stand out. Seeded from what can be known reliably. Voice count comes from counting staves in the slice; key, time and clefs are inherited from the song, because the slices being re-engraved are the illegible ones and reading a key signature off them is exactly the measurement that fails. After the first replacement in a song only the notes need typing. Staff counting needed two corrections against the corpus: compare gaps against line spacing rather than staff height, since adjacent staves can sit closer together than one staff is tall; and require five lines in a group, since Engel's 'uh______' lyric extenders are long horizontal runs too and each counted as a staff. Kaipaava now reads 2,2,2,2,5 on page 1, Ketun 6, Engel 4. Also in this change: - Title is required for export, every other metadata field optional, enforced in bundle.write so the CLI and the editor both get it. Tempo added; noteman already has a free-form column for it. - The panel is a splitter rather than a fixed width, sections collapse under bold grey disclosure headers, and it scrolls. - A re-engraved slice is washed amber with an ENGRAVED badge, and markers get badges too. Thin coloured text was invisible against a scan. Closes #31 Closes #32 Closes #33 Closes #34
250 lines
9.0 KiB
Python
250 lines
9.0 KiB
Python
"""Render project state into finished slice images.
|
||
|
||
load raster → deskew → levels → content rect → cut → discard
|
||
→ trim → scale → pad → ink→alpha → encode
|
||
|
||
The order is not arbitrary. Levels runs before anything geometric so the trim
|
||
bounding box is computed on the image that actually ships; the content
|
||
rectangle runs before cutting so margin junk never enters a slice; and trim
|
||
runs before scale because the scale factor derives from the widest *trimmed*
|
||
slice.
|
||
|
||
Output is final — nothing downstream reprocesses it (ADR 0001).
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
from dataclasses import dataclass
|
||
|
||
import cv2
|
||
import numpy as np
|
||
|
||
from . import lilypond
|
||
from .detect import deskew, staff_height
|
||
from .pdf import Source, page_raster
|
||
from .project import Cut, Project
|
||
|
||
MAX_WIDTH = 1920
|
||
ALPHA_LEVELS = 16 # quantising alpha costs nothing visible and ~32% of the bytes
|
||
_SPECK_AREA = 300 # ink blobs smaller than this don't anchor a trim
|
||
|
||
|
||
@dataclass
|
||
class SliceImage:
|
||
"""One rendered slice, before scaling."""
|
||
|
||
page: int
|
||
index: int
|
||
gray: np.ndarray
|
||
staff: float | None
|
||
|
||
@property
|
||
def width(self) -> int:
|
||
return self.gray.shape[1]
|
||
|
||
|
||
def apply_levels(gray: np.ndarray, black: int, white: int) -> np.ndarray:
|
||
"""Map [black, white] onto the full range with a lookup table.
|
||
|
||
A global LUT, not an adaptive method: CLAHE and adaptive thresholding are
|
||
tuned for text and eat the thin stuff on notation — hairpin tips, slur ends,
|
||
ledger lines, tapered beams.
|
||
"""
|
||
if (black, white) == (0, 255):
|
||
return gray
|
||
lo, hi = min(black, white), max(black, white)
|
||
if hi <= lo:
|
||
return gray
|
||
ramp = np.clip((np.arange(256) - lo) * 255.0 / (hi - lo), 0, 255)
|
||
return cv2.LUT(gray, ramp.astype(np.uint8))
|
||
|
||
|
||
def page_pixels(project: Project, source: Source, index: int) -> np.ndarray:
|
||
"""A page straightened and levelled, ready to be cut."""
|
||
page = project.pages[index]
|
||
gray = deskew(page_raster(source, index), page.skew)
|
||
black, white = project.page_levels(index)
|
||
return apply_levels(gray, black, white)
|
||
|
||
|
||
def _boundary(cut: Cut | None, width: int, height: int, *, bottom: bool) -> list[tuple[int, int]]:
|
||
"""A cut as pixel points spanning the page, or the page edge when absent."""
|
||
if cut is None:
|
||
y = height if bottom else 0
|
||
return [(0, y), (width, y)]
|
||
return [(int(round(x * width)), int(round(y * height))) for x, y in cut.points]
|
||
|
||
|
||
def slice_mask(project: Project, index: int, slot: int, shape: tuple[int, int]) -> np.ndarray:
|
||
"""Which pixels of a page belong to one slice.
|
||
|
||
A slice bounded by a stepped cut is not rectangular, so this is a polygon
|
||
rather than a row range: the top boundary left to right, then the bottom
|
||
boundary right to left.
|
||
"""
|
||
height, width = shape
|
||
page = project.pages[index]
|
||
above, below = page.bounds(slot)
|
||
|
||
polygon = _boundary(above, width, height, bottom=False)
|
||
polygon += _boundary(below, width, height, bottom=True)[::-1]
|
||
|
||
mask = np.zeros(shape, np.uint8)
|
||
cv2.fillPoly(mask, [np.array(polygon, np.int32)], 255)
|
||
|
||
# The content rectangle is applied here rather than as a separate crop, so
|
||
# margin junk can never enter a slice in the first place.
|
||
x0, y0, x1, y1 = project.page_content_rect(index)
|
||
box = np.zeros(shape, np.uint8)
|
||
box[int(y0 * height) : int(y1 * height), int(x0 * width) : int(x1 * width)] = 255
|
||
return cv2.bitwise_and(mask, box)
|
||
|
||
|
||
def _ink_bbox(gray: np.ndarray) -> tuple[int, int, int, int] | None:
|
||
"""Tight bounds of the ink, ignoring specks.
|
||
|
||
One scan fleck at the far left would otherwise anchor the trim and shift
|
||
that slice relative to every other one.
|
||
"""
|
||
ink = (gray < 200).astype(np.uint8)
|
||
count, _, stats, _ = cv2.connectedComponentsWithStats(ink, 8)
|
||
boxes = [
|
||
(
|
||
stats[i, cv2.CC_STAT_LEFT],
|
||
stats[i, cv2.CC_STAT_TOP],
|
||
stats[i, cv2.CC_STAT_LEFT] + stats[i, cv2.CC_STAT_WIDTH],
|
||
stats[i, cv2.CC_STAT_TOP] + stats[i, cv2.CC_STAT_HEIGHT],
|
||
)
|
||
for i in range(1, count)
|
||
if stats[i, cv2.CC_STAT_AREA] >= _SPECK_AREA
|
||
]
|
||
if not boxes:
|
||
return None
|
||
return (
|
||
min(b[0] for b in boxes),
|
||
min(b[1] for b in boxes),
|
||
max(b[2] for b in boxes),
|
||
max(b[3] for b in boxes),
|
||
)
|
||
|
||
|
||
def cut_slice(page: np.ndarray, mask: np.ndarray) -> np.ndarray | None:
|
||
"""Extract one slice: everything outside its region becomes paper.
|
||
|
||
Paper here means white, which the ink→alpha step turns into full
|
||
transparency — so a stepped slice's notch composites invisibly on the
|
||
viewer's sheet rather than covering the neighbouring system.
|
||
"""
|
||
isolated = np.where(mask > 0, page, np.uint8(255))
|
||
box = _ink_bbox(isolated)
|
||
if box is None:
|
||
return None
|
||
x0, y0, x1, y1 = box
|
||
return isolated[y0:y1, x0:x1]
|
||
|
||
|
||
def render_slices(project: Project, source: Source) -> list[SliceImage]:
|
||
"""Every kept slice, trimmed but not yet scaled."""
|
||
out: list[SliceImage] = []
|
||
for index in range(len(project.pages)):
|
||
page_state = project.pages[index]
|
||
# Only rasterize the page if some slice on it still comes from the scan.
|
||
page = None
|
||
for slot in range(page_state.slice_count):
|
||
if page_state.discards[slot]:
|
||
continue
|
||
|
||
engraved = page_state.replacements[slot]
|
||
if engraved and engraved.voices:
|
||
# A re-engraved system enters here, at the trim stage, so it
|
||
# flows through staff-height normalisation and the rest exactly
|
||
# as a scanned one does.
|
||
gray = lilypond.render(
|
||
lilypond.generate(engraved, project.key, project.time)
|
||
)
|
||
else:
|
||
if page is None:
|
||
page = page_pixels(project, source, index)
|
||
gray = cut_slice(page, slice_mask(project, index, slot, page.shape))
|
||
if gray is None:
|
||
continue # a kept slice that turned out to hold no ink
|
||
out.append(SliceImage(index, slot, gray, staff_height(gray, 0, gray.shape[0])))
|
||
return out
|
||
|
||
|
||
def scale_song(slices: list[SliceImage], cap: int = MAX_WIDTH) -> list[np.ndarray]:
|
||
"""Normalise every slice to one staff height, then fit the song to the cap.
|
||
|
||
Two steps, both per song. Staff-height normalisation is what makes a
|
||
rescanned page — or a re-engraved system — sit at the same note size as its
|
||
neighbours; width-based scaling cannot, because width depends on how much
|
||
music is in a system rather than on how big it is drawn.
|
||
|
||
The cap is a ceiling, never a target: a song that comes out narrower stays
|
||
narrower, since enlarging a scan past its own resolution buys softness and
|
||
bytes and no detail.
|
||
"""
|
||
if not slices:
|
||
return []
|
||
|
||
measured = [s.staff for s in slices if s.staff]
|
||
target = float(np.median(measured)) if measured else 0.0
|
||
|
||
factors = [target / s.staff if (target and s.staff) else 1.0 for s in slices]
|
||
widest = max(s.width * f for s, f in zip(slices, factors))
|
||
song = min(1.0, cap / widest) if widest else 1.0
|
||
|
||
out = []
|
||
for s, f in zip(slices, factors):
|
||
k = f * song
|
||
if abs(k - 1.0) < 1e-3:
|
||
out.append(s.gray)
|
||
continue
|
||
interp = cv2.INTER_AREA if k < 1 else cv2.INTER_CUBIC
|
||
out.append(cv2.resize(s.gray, None, fx=k, fy=k, interpolation=interp))
|
||
return out
|
||
|
||
|
||
def pad_right(images: list[np.ndarray]) -> list[np.ndarray]:
|
||
"""Bring every slice to the song's width, flush left.
|
||
|
||
A short system simply ends earlier; the padding is paper, so it disappears
|
||
when ink becomes alpha.
|
||
"""
|
||
if not images:
|
||
return []
|
||
width = max(i.shape[1] for i in images)
|
||
return [
|
||
i
|
||
if i.shape[1] == width
|
||
else cv2.copyMakeBorder(i, 0, 0, 0, width - i.shape[1], cv2.BORDER_CONSTANT, value=255)
|
||
for i in images
|
||
]
|
||
|
||
|
||
def encode(gray: np.ndarray) -> bytes:
|
||
"""Ink black, paper transparent, lossless WebP.
|
||
|
||
Lossless rather than lossy not because lossy looks bad — measured, it
|
||
doesn't — but because it is 58% *larger* on line art (ADR 0003).
|
||
"""
|
||
alpha = 255 - gray
|
||
if ALPHA_LEVELS < 256:
|
||
# Round to the nearest of ALPHA_LEVELS values spanning 0–255 inclusive.
|
||
# Flooring instead would cap full ink at 240 and leave every note
|
||
# slightly transparent.
|
||
step = 255 / (ALPHA_LEVELS - 1)
|
||
alpha = (np.round(alpha / step) * step).astype(np.uint8)
|
||
rgba = np.zeros((*gray.shape, 4), np.uint8)
|
||
rgba[:, :, 3] = alpha
|
||
ok, buf = cv2.imencode(".webp", rgba, [cv2.IMWRITE_WEBP_QUALITY, 101])
|
||
if not ok:
|
||
raise RuntimeError("WebP encoding failed")
|
||
return buf.tobytes()
|
||
|
||
|
||
def render_song(project: Project, source: Source) -> list[bytes]:
|
||
"""The whole raster pipeline: project + PDF in, finished slice images out."""
|
||
slices = render_slices(project, source)
|
||
return [encode(image) for image in pad_right(scale_song(slices))]
|