Re-engraving is a rescue path for the handful of systems a scan cannot deliver, so the window is an editing surface rather than an automation project. Three full-width rows - the scanned system, the render, the form - because a system is wide and short and the job is comparing one against the other bar by bar. The render is shown scaled to the scan's staff height, which is what export does anyway, so it previews the real thing. A form rather than a text box. Key and time are slice-level, clef, notes and lyrics per voice: every staff in a system carries the same key signature, and Kaipaava proves it across five-staff and two-staff systems alike. Notes and lyrics stay raw LilyPond, so slurs, dynamics, tuplets and the laissezVibrer/repeatTie idiom for ties crossing into the next slice all work untouched. Notes are entered in \relative mode, referenced to the middle of each clef's staff, so a part needs no octave marks at all in the common case. The time signature is used for spacing and bar checks but not printed: the printed score repeats the key at every system and the time only at the first, so a re-engraved middle slice showing one would stand out. Seeded from what can be known reliably. Voice count comes from counting staves in the slice; key, time and clefs are inherited from the song, because the slices being re-engraved are the illegible ones and reading a key signature off them is exactly the measurement that fails. After the first replacement in a song only the notes need typing. Staff counting needed two corrections against the corpus: compare gaps against line spacing rather than staff height, since adjacent staves can sit closer together than one staff is tall; and require five lines in a group, since Engel's 'uh______' lyric extenders are long horizontal runs too and each counted as a staff. Kaipaava now reads 2,2,2,2,5 on page 1, Ketun 6, Engel 4. Also in this change: - Title is required for export, every other metadata field optional, enforced in bundle.write so the CLI and the editor both get it. Tempo added; noteman already has a free-form column for it. - The panel is a splitter rather than a fixed width, sections collapse under bold grey disclosure headers, and it scrolls. - A re-engraved slice is washed amber with an ENGRAVED badge, and markers get badges too. Thin coloured text was invisible against a scan. Closes #31 Closes #32 Closes #33 Closes #34
380 lines
15 KiB
Python
380 lines
15 KiB
Python
"""Detection: skew, systems, cuts, staff height.
|
|
|
|
Everything here is a *suggestion* the user confirms or edits (ADR 0004).
|
|
Nothing downstream may assume a result is right.
|
|
|
|
Systems are anchored on the vertical bracket that spans their staves, not on
|
|
gaps in the row-darkness profile: a row profile cannot tell an inter-staff gap
|
|
from an inter-system gap, and gets the count wrong on every page of a
|
|
multi-voice choral score (ADR 0006). The row profile is still needed, to expand
|
|
each anchor to its true ink extent — a bracket stops at the last staff line,
|
|
but the slice must include the lyrics printed below it.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
from dataclasses import dataclass, field
|
|
|
|
import cv2
|
|
import numpy as np
|
|
|
|
SKEW_LIMIT_DEG = 5.0
|
|
SKEW_COARSE_STEP = 1.0
|
|
SKEW_FINE_STEP = 0.1
|
|
_SKEW_WORK_SCALE = 0.25
|
|
|
|
_INK = 128 # below this is ink, above is paper
|
|
_ANCHOR_KERNEL = 0.03 # vertical open kernel, as a fraction of page height
|
|
_ANCHOR_MIN = 0.04 # a bracket is at least this tall, as a fraction of page
|
|
_PROFILE_FLOOR = 0.02 # ink-run threshold, as a fraction of the profile peak
|
|
_EXPAND_REACH = 1.5 # how far past the bracket a system's ink reaches, in staff heights
|
|
_STAFF_KERNEL = 0.05 # horizontal open kernel, as a fraction of page width
|
|
_STAFF_MIN_WIDTH = 0.2 # a staff line spans at least this share of the page
|
|
_CONTENT_MARGIN = 0.01 # slack past the staff ends, for ledger lines and lyrics
|
|
_EDGE_PERCENTILE = 15 # tolerate this share of staff lines merged into scan artefacts
|
|
_STAFF_BREAK = 2.5 # a gap this many line-spacings wide separates two staves
|
|
_STAFF_LINES = 4 # lines a group needs to be a staff rather than an extender (5, minus one for a broken line)
|
|
|
|
|
|
@dataclass
|
|
class Anchor:
|
|
"""A system's vertical bracket: where it is, and how far left it reaches."""
|
|
|
|
top: int
|
|
bottom: int
|
|
left: int
|
|
|
|
|
|
@dataclass
|
|
class System:
|
|
"""One line of music: the ink extent that becomes a slice."""
|
|
|
|
top: int
|
|
bottom: int
|
|
staff_height: float | None = None
|
|
|
|
@property
|
|
def height(self) -> int:
|
|
return self.bottom - self.top
|
|
|
|
|
|
@dataclass
|
|
class PageDetection:
|
|
skew: float
|
|
systems: list[System] = field(default_factory=list)
|
|
cuts: list[int] = field(default_factory=list)
|
|
content: tuple[float, float, float, float] = (0.0, 0.0, 1.0, 1.0)
|
|
|
|
@property
|
|
def bracketless(self) -> bool:
|
|
"""True when no bracket was found and the row profile was used alone."""
|
|
return not self.systems or all(s.staff_height is None for s in self.systems)
|
|
|
|
|
|
def row_darkness(gray: np.ndarray) -> np.ndarray:
|
|
return (255 - gray.astype(np.float32)).sum(axis=1)
|
|
|
|
|
|
def deskew_angle(gray: np.ndarray) -> float:
|
|
"""Angle maximising row-darkness variance — staff lines are the signal.
|
|
|
|
Coarse then fine, on a downscaled copy: 31 warps instead of 101.
|
|
"""
|
|
work = cv2.resize(gray, None, fx=_SKEW_WORK_SCALE, fy=_SKEW_WORK_SCALE,
|
|
interpolation=cv2.INTER_AREA)
|
|
|
|
def score(angle: float) -> float:
|
|
return float(row_darkness(_rotate(work, angle, cv2.INTER_LINEAR)).var())
|
|
|
|
coarse = np.arange(-SKEW_LIMIT_DEG, SKEW_LIMIT_DEG + 1e-9, SKEW_COARSE_STEP)
|
|
best = max(coarse, key=score)
|
|
fine = np.arange(best - SKEW_COARSE_STEP, best + SKEW_COARSE_STEP + 1e-9, SKEW_FINE_STEP)
|
|
fine = fine[np.abs(fine) <= SKEW_LIMIT_DEG]
|
|
return round(float(max(fine, key=score)), 2)
|
|
|
|
|
|
def _rotate(gray: np.ndarray, angle: float, flags: int = cv2.INTER_CUBIC) -> np.ndarray:
|
|
if angle == 0.0:
|
|
return gray
|
|
h, w = gray.shape
|
|
m = cv2.getRotationMatrix2D((w / 2, h / 2), angle, 1.0)
|
|
return cv2.warpAffine(gray, m, (w, h), flags=flags, borderValue=255)
|
|
|
|
|
|
def deskew(gray: np.ndarray, angle: float) -> np.ndarray:
|
|
return _rotate(gray, angle)
|
|
|
|
|
|
def system_anchors(gray: np.ndarray) -> list[Anchor]:
|
|
"""The vertical brackets, one per system."""
|
|
h = gray.shape[0]
|
|
binary = (gray < _INK).astype(np.uint8)
|
|
kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (1, max(3, int(h * _ANCHOR_KERNEL))))
|
|
strokes = cv2.morphologyEx(binary, cv2.MORPH_OPEN, kernel)
|
|
|
|
count, _, stats, _ = cv2.connectedComponentsWithStats(strokes, 8)
|
|
tall = [
|
|
Anchor(
|
|
int(stats[i, cv2.CC_STAT_TOP]),
|
|
int(stats[i, cv2.CC_STAT_TOP] + stats[i, cv2.CC_STAT_HEIGHT]),
|
|
int(stats[i, cv2.CC_STAT_LEFT]),
|
|
)
|
|
for i in range(1, count)
|
|
if stats[i, cv2.CC_STAT_HEIGHT] > h * _ANCHOR_MIN
|
|
]
|
|
|
|
# Tallest first, keeping only strokes that don't overlap one already kept:
|
|
# a system's barlines all overlap its bracket, so each system yields one.
|
|
# The kept stroke is the tallest, which is the bracket rather than a barline.
|
|
anchors: list[Anchor] = []
|
|
for candidate in sorted(tall, key=lambda a: a.bottom - a.top, reverse=True):
|
|
if any(not (candidate.bottom < a.top or candidate.top > a.bottom) for a in anchors):
|
|
continue
|
|
anchors.append(candidate)
|
|
return sorted(anchors, key=lambda a: a.top)
|
|
|
|
|
|
def content_columns(
|
|
gray: np.ndarray, anchors: list[Anchor] | None = None
|
|
) -> tuple[float, float]:
|
|
"""Where the music is horizontally, as normalised x bounds.
|
|
|
|
Staff lines are long *horizontal* runs; a scan-edge shadow, a spine
|
|
darkening and the vertical line a dirty scanner glass leaves are all
|
|
*vertical*. Opening with a wide flat kernel keeps the first and erases the
|
|
others, so the staff lines' own bounding box is the music area.
|
|
|
|
`anchors` does two jobs. It restricts the search to rows known to hold
|
|
systems — without that, a horizontal scan artefact above or below the music
|
|
is itself a long horizontal run reaching the paper edge, which is exactly
|
|
the measurement being avoided. And its brackets give the true left bound:
|
|
a bracket sits *left of every staff line*, so a bound taken from staff
|
|
lines alone crops it off, and a bracket is notation, not artefact.
|
|
|
|
This matters more than it looks: trim is tight and per slice, so one dark
|
|
band down the margin sets that slice's width, which sets the song's widest
|
|
slice, which scales the whole song down.
|
|
"""
|
|
height, width = gray.shape
|
|
binary = (gray < _INK).astype(np.uint8)
|
|
if anchors:
|
|
keep = np.zeros(height, bool)
|
|
for anchor in anchors:
|
|
keep[max(0, anchor.top) : min(height, anchor.bottom)] = True
|
|
binary[~keep] = 0
|
|
kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (max(3, int(width * _STAFF_KERNEL)), 1))
|
|
lines = cv2.morphologyEx(binary, cv2.MORPH_OPEN, kernel)
|
|
|
|
count, _, stats, _ = cv2.connectedComponentsWithStats(lines, 8)
|
|
runs = [
|
|
(stats[i, cv2.CC_STAT_LEFT], stats[i, cv2.CC_STAT_LEFT] + stats[i, cv2.CC_STAT_WIDTH])
|
|
for i in range(1, count)
|
|
if stats[i, cv2.CC_STAT_WIDTH] > width * _STAFF_MIN_WIDTH
|
|
]
|
|
if not runs:
|
|
return 0.0, 1.0
|
|
|
|
# Percentiles, not the extremes. Where a scan-edge band happens to touch
|
|
# the end of a staff line the two merge into one component, and that
|
|
# component then reaches into the artefact — on Ketun joululaulu p2 the
|
|
# merged line ends at 1575px against 1544px on the clean page. A page has
|
|
# dozens of staff lines and only a few are contaminated, so a percentile
|
|
# lands on the true edge while the extreme lands on the worst artefact.
|
|
lefts = np.array([r[0] for r in runs], float)
|
|
rights = np.array([r[1] for r in runs], float)
|
|
margin = width * _CONTENT_MARGIN
|
|
|
|
left = float(np.percentile(lefts, _EDGE_PERCENTILE))
|
|
if anchors:
|
|
left = min(left, min(a.left for a in anchors))
|
|
right = float(np.percentile(rights, 100 - _EDGE_PERCENTILE))
|
|
|
|
return max(0.0, left - margin) / width, min(float(width), right + margin) / width
|
|
|
|
|
|
def staff_count(gray: np.ndarray) -> int:
|
|
"""How many staves are in this slice — i.e. how many voices it holds.
|
|
|
|
Kaipaava's first four systems have two staves and its fifth has five, so
|
|
this cannot be a song-level constant. Counts long horizontal runs and
|
|
divides by the five lines a staff has; the same signal that finds the music
|
|
area, so it degrades the same way and no worse.
|
|
"""
|
|
height, width = gray.shape
|
|
binary = (gray < _INK).astype(np.uint8)
|
|
kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (max(3, int(width * _STAFF_KERNEL)), 1))
|
|
lines = cv2.morphologyEx(binary, cv2.MORPH_OPEN, kernel)
|
|
|
|
count, _, stats, _ = cv2.connectedComponentsWithStats(lines, 8)
|
|
rows = sorted(
|
|
stats[i, cv2.CC_STAT_TOP]
|
|
for i in range(1, count)
|
|
if stats[i, cv2.CC_STAT_WIDTH] > width * _STAFF_MIN_WIDTH
|
|
)
|
|
if not rows:
|
|
return 1
|
|
|
|
# Compare against the *line* spacing, not the staff height: adjacent staves
|
|
# can sit closer together than one staff is tall, so a staff-height
|
|
# threshold merges them into one.
|
|
line_spacing = (staff_height(gray, 0, height) or height * 0.05) / 4
|
|
|
|
groups: list[list[int]] = [[rows[0]]]
|
|
for row in rows[1:]:
|
|
if row - groups[-1][-1] > line_spacing * _STAFF_BREAK:
|
|
groups.append([])
|
|
groups[-1].append(row)
|
|
|
|
# A staff is five evenly spaced lines. Lone long runs are lyric extenders —
|
|
# Engel's "uh______" — and hairpins, which are just as horizontal as a
|
|
# staff line and would otherwise each count as a staff.
|
|
staves = sum(1 for group in groups if len(group) >= _STAFF_LINES)
|
|
return max(1, staves)
|
|
|
|
|
|
def ink_runs(gray: np.ndarray) -> list[tuple[int, int]]:
|
|
"""Rows containing ink, despeckled — specks are the known failure mode."""
|
|
profile = row_darkness(cv2.medianBlur(gray, 3))
|
|
if profile.max() <= 0:
|
|
return []
|
|
inked = profile > profile.max() * _PROFILE_FLOOR
|
|
|
|
runs: list[tuple[int, int]] = []
|
|
start: int | None = None
|
|
for i, on in enumerate(inked):
|
|
if on and start is None:
|
|
start = i
|
|
elif not on and start is not None:
|
|
runs.append((start, i))
|
|
start = None
|
|
if start is not None:
|
|
runs.append((start, len(inked)))
|
|
return runs
|
|
|
|
|
|
def staff_height(gray: np.ndarray, top: int, bottom: int) -> float | None:
|
|
"""Distance between a staff's outer lines, from staff-line spacing."""
|
|
profile = row_darkness(gray[top:bottom])
|
|
if profile.size == 0 or profile.max() <= 0:
|
|
return None
|
|
peaks = np.where(profile > profile.max() * 0.55)[0]
|
|
if peaks.size < 2:
|
|
return None
|
|
|
|
centres = []
|
|
run = [peaks[0]]
|
|
for prev, cur in zip(peaks, peaks[1:]):
|
|
if cur - prev > 3:
|
|
centres.append(float(np.mean(run)))
|
|
run = []
|
|
run.append(cur)
|
|
centres.append(float(np.mean(run)))
|
|
if len(centres) < 2:
|
|
return None
|
|
|
|
gaps = np.diff(centres)
|
|
# Keep intra-staff gaps; the big ones are the spaces between staves.
|
|
intra = gaps[gaps < np.median(gaps) * 2]
|
|
if intra.size == 0:
|
|
return None
|
|
return float(np.median(intra) * 4) # 5 lines, 4 spaces
|
|
|
|
|
|
def _gap(run: tuple[int, int], anchor: Anchor) -> int:
|
|
"""Vertical distance between an ink run and a bracket; 0 if they overlap."""
|
|
start, end = run
|
|
if end > anchor.top and start < anchor.bottom:
|
|
return 0
|
|
return anchor.top - end if end <= anchor.top else start - anchor.bottom
|
|
|
|
|
|
def _assign(
|
|
runs: list[tuple[int, int]],
|
|
anchors: list[Anchor],
|
|
reaches: list[float],
|
|
) -> list[tuple[int, int]]:
|
|
"""Give every ink run to one system, and return each system's extent.
|
|
|
|
A run between two systems is resolved by **precedence, not proximity**: the
|
|
system above wins if the run is within its reach. Text printed under a staff
|
|
belongs to that staff, and engravers space lyrics generously — on *Feliz
|
|
Navidad* a lyric line sits 43px under its own system's bracket but only 10px
|
|
above the next one's, so nearest-bracket gives it to the wrong system.
|
|
|
|
Distance is measured from the *bracket*, never from a growing extent — a
|
|
title block's credit lines are stacked closely enough that a chaining
|
|
expansion hops from one to the next and walks the whole way up the page.
|
|
|
|
One pass over all systems, rather than each bracket expanding on its own, so
|
|
that a run has exactly one owner and extents cannot overlap.
|
|
|
|
Known limit: when a lyric line is printed tight enough under its system that
|
|
no blank row separates it from the *next* system's staves, the two fuse into
|
|
a single ink run and no row profile can split them — the lyric is then given
|
|
to the system below and the cut lands high. Dragging the cut is the fix;
|
|
separating them needs a signal this pass doesn't have.
|
|
"""
|
|
bounds = [[a.top, a.bottom] for a in anchors]
|
|
|
|
def claim(index: int, run: tuple[int, int]) -> None:
|
|
bounds[index][0] = min(bounds[index][0], run[0])
|
|
bounds[index][1] = max(bounds[index][1], run[1])
|
|
|
|
for run in runs:
|
|
gaps = [_gap(run, a) for a in anchors]
|
|
|
|
# Ink overlapping a bracket belongs to it — to the one it overlaps most,
|
|
# whatever else is in reach.
|
|
inside = [
|
|
(min(run[1], anchors[i].bottom) - max(run[0], anchors[i].top), i)
|
|
for i, g in enumerate(gaps)
|
|
if g == 0
|
|
]
|
|
if inside:
|
|
claim(max(inside)[1], run)
|
|
continue
|
|
|
|
within = [i for i, g in enumerate(gaps) if g <= reaches[i]]
|
|
if not within:
|
|
continue # a title block or a footer: too far from any system
|
|
|
|
# Otherwise the system above wins, and only failing that the one below.
|
|
above = [i for i in within if anchors[i].bottom <= run[0]]
|
|
claim(above[-1] if above else within[0], run)
|
|
|
|
return [(lo, hi) for lo, hi in bounds]
|
|
|
|
|
|
def detect_page(gray: np.ndarray, skew: float | None = None) -> PageDetection:
|
|
"""Full proposal for one page raster. `gray` is the *unrotated* page."""
|
|
angle = deskew_angle(gray) if skew is None else skew
|
|
straight = deskew(gray, angle)
|
|
|
|
runs = ink_runs(straight)
|
|
anchors = system_anchors(straight)
|
|
|
|
if anchors:
|
|
# Staff height is measured on the bracket span, before expansion, so a
|
|
# swallowed title block can't distort it.
|
|
heights = [staff_height(straight, a.top, a.bottom) for a in anchors]
|
|
reaches = [(h or gray.shape[0] * 0.02) * _EXPAND_REACH for h in heights]
|
|
systems = [
|
|
System(top=lo, bottom=hi, staff_height=h)
|
|
for (lo, hi), h in zip(_assign(runs, anchors, reaches), heights)
|
|
]
|
|
else:
|
|
# No bracket: a single-staff melody or lead sheet, where every ink run
|
|
# genuinely is its own system.
|
|
systems = [System(top=t, bottom=b) for t, b in runs]
|
|
|
|
cuts = [
|
|
(systems[i].bottom + systems[i + 1].top) // 2 for i in range(len(systems) - 1)
|
|
]
|
|
# Only the horizontal bounds are proposed. Vertically the cuts and the
|
|
# discard flags already isolate the header and footer, and cropping the top
|
|
# would risk clipping a tempo mark or a section label above the first staff.
|
|
left, right = content_columns(straight, anchors)
|
|
return PageDetection(
|
|
skew=angle, systems=systems, cuts=cuts, content=(left, 0.0, right, 1.0)
|
|
)
|