diff --git a/noteman_slicer/cli.py b/noteman_slicer/cli.py index 2aaa345..2f804ff 100644 --- a/noteman_slicer/cli.py +++ b/noteman_slicer/cli.py @@ -50,6 +50,38 @@ def _detect(args: argparse.Namespace) -> int: return 0 +def _project(args: argparse.Namespace) -> int: + from .project import Project, default_path + + source = open_source(args.pdf, SourceType(args.type) if args.type else None) + path = default_path(source.path) + + if path.exists() and not args.force: + project = Project.load(path) + print(f"{path.name}: loaded") + if project.source_changed(): + print(" WARNING: the PDF has changed since these cuts were made") + else: + detections, heights = [], [] + for i in range(len(source)): + gray = page_raster(source, i) + detections.append(detect_page(gray)) + heights.append(gray.shape[0]) + project = Project.from_detection(source.path, detections, heights) + print(f"{path.name}: created from detection") + + kept = project.kept_slices() + for i, page in enumerate(project.pages): + flags = "".join("." if d else "#" for d in page.discards) + print(f" p{i + 1:<3} skew {page.skew:+.2f}° {page.slice_count} slices [{flags}]") + print(f" {len(kept)} slices kept, {sum(p.slice_count for p in project.pages) - len(kept)} discarded") + + if args.save: + print(f" saved to {project.save(path)}") + source.close() + return 0 + + def main(argv: list[str] | None = None) -> int: parser = argparse.ArgumentParser( prog="noteman-slicer", @@ -74,6 +106,13 @@ def main(argv: list[str] | None = None) -> int: det.add_argument("--type", choices=[t.value for t in SourceType]) det.set_defaults(func=_detect) + proj = sub.add_parser("project", help="create or inspect the project file for a PDF") + proj.add_argument("pdf") + proj.add_argument("--save", action="store_true", help="write the project file") + proj.add_argument("--force", action="store_true", help="re-detect, discarding existing state") + proj.add_argument("--type", choices=[t.value for t in SourceType]) + proj.set_defaults(func=_project) + args = parser.parse_args(argv) return args.func(args) diff --git a/noteman_slicer/project.py b/noteman_slicer/project.py new file mode 100644 index 0000000..3375cca --- /dev/null +++ b/noteman_slicer/project.py @@ -0,0 +1,241 @@ +"""Project state: everything the human decided, on disk beside the PDF. + +The bundle is generated from this, so export is a pure function of the project +plus the PDF. That buys crash safety, resume across sessions, and re-export — +change the width cap or fix one cut and every song regenerates without +repeating any human work. + +All geometry is stored in **normalised page coordinates** (0–1 of the deskewed +page), so the file is independent of DPI and of which renderer produced it. +""" + +from __future__ import annotations + +import hashlib +import json +from dataclasses import dataclass, field +from pathlib import Path + +from .detect import PageDetection + +FORMAT_VERSION = 1 +SUFFIX = ".slicer.json" + +Point = tuple[float, float] + + +@dataclass +class Cut: + """A boundary splitting one slice into two, spanning the page left to right. + + A polyline, not a line. Two points is the ordinary straight case; extra + vertices handle a section label printed in the left margin at the same + height as the previous system's lyrics, where no horizontal line separates + the two (see docs/spec.md). + """ + + points: list[Point] + + @classmethod + def straight(cls, y: float) -> Cut: + return cls([(0.0, y), (1.0, y)]) + + @property + def straight_y(self) -> float | None: + """The single y of a straight cut, or None if it steps.""" + ys = {y for _, y in self.points} + return self.points[0][1] if len(ys) == 1 else None + + def y_at(self, x: float) -> float: + """Height of the boundary at a horizontal position.""" + pts = self.points + if x <= pts[0][0]: + return pts[0][1] + for (x0, y0), (x1, y1) in zip(pts, pts[1:]): + if x <= x1: + if x1 == x0: + return y1 + return y0 + (y1 - y0) * (x - x0) / (x1 - x0) + return pts[-1][1] + + @property + def lowest(self) -> float: + return max(y for _, y in self.points) + + @property + def highest(self) -> float: + return min(y for _, y in self.points) + + +@dataclass +class Page: + """One page's decisions. `cuts` are ordered top to bottom.""" + + skew: float = 0.0 + cuts: list[Cut] = field(default_factory=list) + discards: list[bool] = field(default_factory=lambda: [False]) + content_rect: tuple[float, float, float, float] | None = None + levels: tuple[int, int] | None = None + + @property + def slice_count(self) -> int: + return len(self.cuts) + 1 + + def bounds(self, index: int) -> tuple[Cut | None, Cut | None]: + """The cuts above and below a slice; None means the page edge.""" + above = self.cuts[index - 1] if index > 0 else None + below = self.cuts[index] if index < len(self.cuts) else None + return above, below + + def add_cut(self, cut: Cut) -> int: + """Insert a cut, splitting the slice it lands in. Returns its index.""" + y = cut.points[0][1] + index = sum(1 for c in self.cuts if c.points[0][1] < y) + self.cuts.insert(index, cut) + # The split slice keeps its flag on both halves. + self.discards.insert(index, self.discards[index]) + return index + + def remove_cut(self, index: int) -> None: + """Drop a cut, merging the two slices it separated.""" + self.cuts.pop(index) + merged = self.discards[index] and self.discards[index + 1] + self.discards.pop(index + 1) + self.discards[index] = merged + + +@dataclass +class Project: + source: Path + source_hash: str + pages: list[Page] + content_rect: tuple[float, float, float, float] = (0.0, 0.0, 1.0, 1.0) + levels: tuple[int, int] = (0, 255) + metadata: dict[str, str] = field(default_factory=dict) + path: Path | None = None + + # -- geometry helpers ------------------------------------------------- + + def page_content_rect(self, index: int) -> tuple[float, float, float, float]: + return self.pages[index].content_rect or self.content_rect + + def page_levels(self, index: int) -> tuple[int, int]: + return self.pages[index].levels or self.levels + + def kept_slices(self) -> list[tuple[int, int]]: + """(page, slice) of every slice that will be exported, in song order.""" + return [ + (p, s) + for p, page in enumerate(self.pages) + for s in range(page.slice_count) + if not page.discards[s] + ] + + # -- persistence ------------------------------------------------------ + + @classmethod + def from_detection( + cls, source: Path, detections: list[PageDetection], heights: list[int] + ) -> Project: + """Seed a project from detection. Every value here is a suggestion. + + Detection emits cuts only *between* systems, so a page would otherwise + have exactly as many slices as it has systems, with the header and + footer inside the first and last. The boundary cuts that isolate them — + and the discard flags that drop them — are a slicing decision, not a + detection result, so they are added here. + """ + pages = [] + for detection, height in zip(detections, heights): + ys = list(detection.cuts) + leading = trailing = False + + if detection.systems: + first, last = detection.systems[0], detection.systems[-1] + if first.top > 0: + ys.insert(0, first.top // 2) + leading = True + if last.bottom < height: + ys.append((last.bottom + height) // 2) + trailing = True + + discards = [False] * (len(ys) + 1) + if leading: + discards[0] = True + if trailing: + discards[-1] = True + + pages.append( + Page( + skew=detection.skew, + cuts=[Cut.straight(y / height) for y in ys], + discards=discards, + ) + ) + return cls(source=source, source_hash=hash_file(source), pages=pages) + + def save(self, path: Path | None = None) -> Path: + """Atomic write, so a crash mid-save cannot destroy the previous state.""" + target = Path(path or self.path or default_path(self.source)) + payload = { + "v": FORMAT_VERSION, + "source": self.source.name, + "source_hash": self.source_hash, + "content_rect": list(self.content_rect), + "levels": list(self.levels), + "metadata": self.metadata, + "pages": [ + { + "skew": page.skew, + "cuts": [[list(p) for p in cut.points] for cut in page.cuts], + "discards": page.discards, + "content_rect": list(page.content_rect) if page.content_rect else None, + "levels": list(page.levels) if page.levels else None, + } + for page in self.pages + ], + } + tmp = target.with_suffix(target.suffix + ".tmp") + tmp.write_text(json.dumps(payload, indent=2, ensure_ascii=False)) + tmp.replace(target) + self.path = target + return target + + @classmethod + def load(cls, path: Path, source: Path | None = None) -> Project: + path = Path(path) + data = json.loads(path.read_text()) + if data.get("v") != FORMAT_VERSION: + raise ValueError(f"unsupported project version {data.get('v')!r}") + pdf = Path(source) if source else path.parent / data["source"] + pages = [ + Page( + skew=page["skew"], + cuts=[Cut([tuple(p) for p in cut]) for cut in page["cuts"]], + discards=page["discards"], + content_rect=tuple(page["content_rect"]) if page["content_rect"] else None, + levels=tuple(page["levels"]) if page["levels"] else None, + ) + for page in data["pages"] + ] + return cls( + source=pdf, + source_hash=data["source_hash"], + pages=pages, + content_rect=tuple(data["content_rect"]), + levels=tuple(data["levels"]), + metadata=data.get("metadata", {}), + path=path, + ) + + def source_changed(self) -> bool: + """True when the PDF no longer matches what these decisions were made on.""" + return self.source.exists() and hash_file(self.source) != self.source_hash + + +def default_path(source: Path) -> Path: + return Path(source).with_suffix(SUFFIX) + + +def hash_file(path: Path) -> str: + return hashlib.sha256(Path(path).read_bytes()).hexdigest() diff --git a/tests/test_project.py b/tests/test_project.py new file mode 100644 index 0000000..9c93497 --- /dev/null +++ b/tests/test_project.py @@ -0,0 +1,82 @@ +"""Runnable check for project state: round-trip, cut edits, discard pre-set.""" + +from __future__ import annotations + +import sys +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parents[1])) + +from noteman_slicer.detect import PageDetection, System # noqa: E402 +from noteman_slicer.project import Cut, Project, default_path # noqa: E402 + + +def main() -> int: + tmp = Path(__file__).with_name("_tmp") + tmp.mkdir(exist_ok=True) + pdf = tmp / "song.pdf" + pdf.write_bytes(b"%PDF-1.7 not really a pdf, only its bytes are hashed") + + height = 1000 + detection = PageDetection( + skew=-1.1, + systems=[System(200, 400, 60.0), System(600, 800, 60.0)], + cuts=[500], + ) + project = Project.from_detection(pdf, [detection], [height]) + page = project.pages[0] + + # One cut between the systems, plus a boundary cut above the first and + # below the last — so the header and footer become their own slices. + assert len(page.cuts) == 3, [c.points for c in page.cuts] + assert page.discards == [True, False, False, True], page.discards + assert page.slice_count == 4 + assert project.kept_slices() == [(0, 1), (0, 2)], project.kept_slices() + + # Geometry is normalised, so it survives any change of resolution. + assert all(0.0 <= y <= 1.0 for cut in page.cuts for _, y in cut.points) + assert page.cuts[1].straight_y == 0.5 + + # A straight cut is flat; a stepped one is not, and interpolates. + step = Cut([(0.0, 0.20), (0.35, 0.20), (0.35, 0.40), (1.0, 0.40)]) + assert step.straight_y is None + assert step.y_at(0.0) == 0.20 + assert step.y_at(1.0) == 0.40 + assert step.y_at(0.35) == 0.20 or step.y_at(0.35) == 0.40 + assert step.highest == 0.20 and step.lowest == 0.40 + + # Adding a cut splits a slice and keeps that slice's flag on both halves. + before = page.slice_count + index = page.add_cut(Cut.straight(0.65)) + assert page.slice_count == before + 1 + assert index == 2, index + assert page.discards == [True, False, False, False, True], page.discards + + # Removing it merges them again. + page.remove_cut(index) + assert page.slice_count == before + assert page.discards == [True, False, False, True], page.discards + + # Round-trip. + saved = project.save() + assert saved == default_path(pdf), saved + reloaded = Project.load(saved) + assert reloaded.pages[0].skew == -1.1 + assert reloaded.pages[0].discards == page.discards + assert [c.points for c in reloaded.pages[0].cuts] == [c.points for c in page.cuts] + assert reloaded.source_hash == project.source_hash + assert not reloaded.source_changed() + + # A PDF edited underneath must be reported, not silently re-cut. + pdf.write_bytes(b"%PDF-1.7 different bytes entirely") + assert reloaded.source_changed() + + for f in (pdf, saved): + f.unlink() + tmp.rmdir() + print("ok") + return 0 + + +if __name__ == "__main__": + sys.exit(main())