Scanned scores are black ink on white paper stored as 8-bit greyscale or RGB, which costs several times what the same page costs as a bilevel image. Across an 11-song corpus this is 27.6 MB to 7.7 MB; Engel's bundle goes from 5997 KB to 2092 KB with byte-identical slices, since only the archived copy changes. Three approaches were measured and discarded first, which is worth recording because two of them are the obvious ones. Converting RGB to greyscale and re-encoding makes these files 20-86% LARGER: the source JPEGs are already near 0.7 bits per pixel, so re-encoding adds generation loss and spends more bits than the original did, and dropping chroma recovers nothing because JPEG already subsamples it. Lossless structural optimisation gains 0.1%, because images are 99% of every file and there are no duplicates. Downsampling works but 300 DPI is print resolution, and the PDF exists to be printed. Two failure modes were found by looking at output rather than at byte counts, and both are now refused: - A scan at ~115 DPI came back with broken staff lines. Guarded on resolution as the image is *placed on the page*, so a tiled scan with 126 small images still qualifies where a pixel count would reject it. - Cover artwork was flattened to grey. Guarded on chroma: artwork measures 44% off-grey against 3% for sensor tint on a greyscale scan. The first threshold of 2% was a false positive that cost 685 KB on one song for nothing; 10% sits in the gap with room either side. Exposed as a button rather than a checkbox. It reports what it skipped and why, and shows a before/after crop, because the failure it can produce is obvious at a glance and invisible in a size figure. Off by default: this is lossy on the copy kept for printing.
131 lines
4.4 KiB
Python
131 lines
4.4 KiB
Python
"""Bundle export — the only channel to noteman (ADR 0001).
|
|
|
|
song.zip
|
|
song.json
|
|
original.pdf
|
|
001.webp 002.webp …
|
|
|
|
Array order in `song.json` *is* slice order: one ordering, not two. Markers
|
|
nest inside the slice they sit on, so an index appears in exactly one place —
|
|
a jump source's `destination`.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import zipfile
|
|
from pathlib import Path
|
|
|
|
from .pdf import Source
|
|
from .project import Project
|
|
from .render import render_song
|
|
|
|
FORMAT_VERSION = 1
|
|
METADATA_FIELDS = (
|
|
"title",
|
|
"subtitle",
|
|
"composer",
|
|
"original_artist",
|
|
"arranger",
|
|
"lyricist",
|
|
"translator",
|
|
"tempo",
|
|
"voices",
|
|
)
|
|
|
|
# Beats per minute, exported as a JSON number. A figure is worth more than a
|
|
# word here: "Andante" cannot drive a metronome and two people will not agree
|
|
# what it means.
|
|
NUMERIC_FIELDS = frozenset({"tempo"})
|
|
|
|
|
|
def song_json(project: Project, files: list[str]) -> dict:
|
|
payload: dict = {"v": FORMAT_VERSION}
|
|
for field in METADATA_FIELDS:
|
|
value = (project.metadata.get(field) or "").strip()
|
|
if not value:
|
|
continue
|
|
if field in NUMERIC_FIELDS:
|
|
try:
|
|
payload[field] = int(value)
|
|
except ValueError:
|
|
continue # not a number, so not worth exporting as one
|
|
else:
|
|
payload[field] = value
|
|
|
|
kept = project.kept_slices()
|
|
# Markers reference slices by (page, slot) while editing, because that is
|
|
# what survives adding and removing cuts. In the bundle they become the
|
|
# array index, which is the only cross-reference the format has.
|
|
index_of = {position: i for i, position in enumerate(kept)}
|
|
|
|
slices: list[dict] = []
|
|
for name, (page, slot) in zip(files, kept):
|
|
entry: dict = {"file": name}
|
|
markers = []
|
|
for marker in project.pages[page].markers[slot]:
|
|
item: dict = {"type": marker.type}
|
|
if marker.label:
|
|
item["label"] = marker.label
|
|
if marker.destination is not None:
|
|
target = index_of.get(tuple(marker.destination))
|
|
# A jump whose target was discarded or re-cut away is dropped
|
|
# rather than exported dangling: noteman would have nothing to
|
|
# resolve it to.
|
|
if target is None:
|
|
continue
|
|
item["destination"] = target
|
|
markers.append(item)
|
|
if markers:
|
|
entry["markers"] = markers
|
|
slices.append(entry)
|
|
|
|
payload["slices"] = slices
|
|
return payload
|
|
|
|
|
|
def write(project: Project, source: Source, path: Path) -> Path:
|
|
"""Render the song and write the bundle. Returns the zip path.
|
|
|
|
A title is required; every other metadata field is optional. noteman's own
|
|
rule is that a song needs a title and at least one slice, and a bundle that
|
|
cannot become a song is not worth writing.
|
|
"""
|
|
if not project.metadata.get("title", "").strip():
|
|
raise ValueError("a title is required before a song can be exported")
|
|
|
|
images = render_song(project, source)
|
|
if not images:
|
|
raise ValueError("no slices to export — every slice is discarded")
|
|
names = [f"{i + 1:03}.webp" for i in range(len(images))]
|
|
|
|
path = Path(path)
|
|
path.parent.mkdir(parents=True, exist_ok=True)
|
|
# ZIP_STORED for the images: WebP is already compressed, so deflating it
|
|
# only costs time. The JSON is small enough not to care.
|
|
with zipfile.ZipFile(path, "w") as zf:
|
|
zf.writestr(
|
|
"song.json",
|
|
json.dumps(song_json(project, names), indent=2, ensure_ascii=False),
|
|
zipfile.ZIP_DEFLATED,
|
|
)
|
|
if project.source.exists():
|
|
pdf = project.source.read_bytes()
|
|
if project.optimise_pdf:
|
|
import pymupdf
|
|
|
|
from .pdfopt import optimise
|
|
|
|
shrunk, _ = optimise(pymupdf.open(project.source), len(pdf))
|
|
pdf = shrunk or pdf # empty means it found no saving
|
|
zf.writestr("original.pdf", pdf, zipfile.ZIP_STORED)
|
|
for name, data in zip(names, images):
|
|
zf.writestr(name, data, zipfile.ZIP_STORED)
|
|
|
|
# The project is spent once its song has been exported: the next edit
|
|
# session starts fresh from detection rather than resuming these decisions.
|
|
# Recorded here so no caller can forget it.
|
|
project.exported = True
|
|
project.save()
|
|
return path
|