from __future__ import annotations import argparse import threading import tkinter as tk from tkinter import messagebox from collections.abc import Sequence from functools import lru_cache from pathlib import Path from queue import Empty, Queue from threading import Thread from pdf2image import convert_from_path from PIL import Image, ImageTk from copienator import ( CliError, EvaluationWorkspace, ExitCode, atomic_write_json, execute, target_parser, workspace_from_target, ) from copienator.filesystem import staged_files from copienator.copy_errors import clear_copy_error, mark_copy_error, marked_copy_paths DELIMITER_WIDTH = 5 DELIMITER_COLOR = (0, 0, 0) OUTPUT_SIZE = (1800, 1000) pdf_cache_lock = threading.Lock() def distribute_pages(total_pages: int, max_per_file: int = 5) -> list[int]: """Distribute pages into balanced chunks no larger than max_per_file.""" if total_pages == 0: return [] number_of_files = (total_pages + max_per_file - 1) // max_per_file base_count, remainder = divmod(total_pages, number_of_files) return [ base_count + (1 if index < remainder else 0) for index in range(number_of_files) ] def stitch_images(image_list: list[Image.Image]) -> Image.Image | None: if not image_list: return None total_width = sum(image.width for image in image_list) total_width += (len(image_list) - 1) * DELIMITER_WIDTH max_height = max(image.height for image in image_list) combined = Image.new("RGB", (total_width, max_height), color="white") x_offset = 0 for index, image in enumerate(image_list): combined.paste(image, (x_offset, 0)) x_offset += image.width if index < len(image_list) - 1: delimiter = Image.new( "RGB", (DELIMITER_WIDTH, max_height), color=DELIMITER_COLOR ) combined.paste(delimiter, (x_offset, 0)) x_offset += DELIMITER_WIDTH return combined @lru_cache(maxsize=3) def _get_pdf_pages_cached(pdf_path: Path) -> list[Image.Image]: # Label coordinates use the full, displayed MediaBox, including on PDFs # previously cropped by crop-margins. Keep this in sync with split-answers. return convert_from_path(pdf_path, use_cropbox=False) def get_pdf_pages(pdf_path: Path) -> list[Image.Image]: """Thread-safe wrapper around the small PDF conversion cache.""" with pdf_cache_lock: return _get_pdf_pages_cached(pdf_path) def process_single_pdf( pdf_path: Path, shift_offset: int = 0, max_per_file: int = 5, ) -> tuple[Image.Image, list[Image.Image], dict[str, object]] | None: """Convert one PDF into a preview, full-resolution splits and metadata.""" try: cropped_images = [] for image in get_pdf_pages(pdf_path): width, height = image.size if max_per_file == 1: left, right = 0, width else: left = max(0, 100 + shift_offset) right = min(width, width // 3 + 100 + shift_offset) if right > left: cropped_images.append(image.crop((left, 0, right, height))) if not cropped_images: return None distribution = distribute_pages(len(cropped_images), max_per_file) split_images = [] current_index = 0 for count in distribution: stitched = stitch_images(cropped_images[current_index : current_index + count]) if stitched is not None: split_images.append(stitched) current_index += count full_stitch = stitch_images(cropped_images) if full_stitch is None: return None preview = full_stitch.resize(OUTPUT_SIZE, Image.Resampling.BILINEAR) schema: dict[str, object] = { "original_filename": pdf_path.name, "total_pages": len(cropped_images), "number_of_files": len(split_images), "columns_per_file": distribution, } return preview, split_images, schema except Exception as exc: # noqa: BLE001 - interactive item failure print(f"Error processing {pdf_path.name}: {exc}") return None def _previous_cutleft_outputs(output_dir: Path, base_name: str) -> set[str]: if not output_dir.is_dir(): return set() result = {f"{base_name}_schema.json"} for path in output_dir.glob(f"{base_name}_*.jpg"): suffix = path.stem.removeprefix(f"{base_name}_") if suffix.isdigit(): result.add(path.name) return result def save_results( result: tuple[Image.Image, list[Image.Image], dict[str, object]], pdf_path: Path, output_dir: Path, ) -> None: """Atomically replace every Cutleft output associated with one copy.""" _, splits, schema = result base_name = pdf_path.stem previous = _previous_cutleft_outputs(output_dir, base_name) with staged_files(output_dir, remove=previous) as staging: for index, image in enumerate(splits, start=1): filename = f"{base_name}_{index:02d}.jpg" image.save(staging / filename, "JPEG", quality=95) atomic_write_json(staging / f"{base_name}_schema.json", schema) for index in range(1, len(splits) + 1): print(f"Saved: {base_name}_{index:02d}.jpg") print(f"Saved schema: {base_name}_schema.json") class ImageReviewer: def __init__( self, files: list[Path], output_dir: Path, default_max_per_file: int = 5, ) -> None: self.files = files self.output_dir = output_dir self.workspace = EvaluationWorkspace(output_dir.parent) self.completed = False self.had_errors = False self.stop_prefetch = threading.Event() self.current_result = None self.index = 0 self.current_shift = 0 self.default_max_per_file = default_max_per_file self.current_max_per_file = default_max_per_file self.current_preview: Image.Image | None = None self.is_processing = False self.manual_queue: Queue[ tuple[Image.Image, list[Image.Image], dict[str, object]] | None ] = Queue() self.root = tk.Tk() self.root.title("PDF Cropper") self.root.geometry("+100+100") self.label_img = tk.Label(self.root) self.label_img.pack() self.label_info = tk.Label(self.root, text="", font=("Arial", 12, "bold")) self.label_info.pack(pady=5) self.root.bind("", self.on_next) self.root.bind("s", self.on_skip) self.root.protocol("WM_DELETE_WINDOW", self.on_close) self.root.bind("n", lambda _event: self.on_shift(50)) self.root.bind("N", lambda _event: self.on_shift(100)) self.root.bind("t", lambda _event: self.on_shift(-50)) self.root.bind("1", lambda _event: self.on_set_max_pages(1)) Thread(target=self.prefetch_worker, daemon=True).start() self.load_current_image() self.root.lift() self.root.focus_force() self.root.mainloop() def on_set_max_pages(self, count: int) -> None: if self.is_processing: return self.current_max_per_file = count print(f"Setting max pages per file: {count}") self.trigger_processing(self.files[self.index], self.current_shift) def prefetch_worker(self) -> None: processed_index = -1 while not self.stop_prefetch.is_set(): target = self.index + 1 if target < len(self.files) and target != processed_index: try: get_pdf_pages(self.files[target]) except Exception: pass # The foreground review reports and flags conversion errors. processed_index = target self.stop_prefetch.wait(0.05) def load_current_image(self) -> None: if self.index >= len(self.files): print("All files processed.") self.completed = True self.on_close() return self.is_processing = False self.current_shift = 0 self.current_result = None self.trigger_processing(self.files[self.index], self.current_shift) def trigger_processing(self, pdf_path: Path, shift: int) -> None: self.is_processing = True self.label_info.configure( text=f"Processing {pdf_path.name} (Shift {shift})... Please wait.", fg="red", ) def worker() -> None: self.manual_queue.put( process_single_pdf(pdf_path, shift, self.current_max_per_file) ) Thread(target=worker, daemon=True).start() self.check_manual_queue(pdf_path) def check_manual_queue(self, pdf_path: Path) -> None: try: result = self.manual_queue.get_nowait() self.is_processing = False if result is None: print(f"Failed to process {pdf_path.name}, skipping.") self._mark_error(pdf_path, "Échec de la conversion pour la découpe des marges") self._advance() else: self.handle_processing_result(result, pdf_path) except Empty: self.root.after(100, lambda: self.check_manual_queue(pdf_path)) def handle_processing_result( self, result: tuple[Image.Image, list[Image.Image], dict[str, object]], pdf_path: Path, ) -> None: self.current_preview = result[0] self.current_result = result self.update_display(pdf_path.name, result[2]) def update_display(self, filename: str, schema: dict[str, object]) -> None: if self.current_preview is None: return tk_image = ImageTk.PhotoImage(self.current_preview) self.label_img.configure(image=tk_image) self.label_img.image = tk_image self.label_info.configure( text=( f"[{self.index + 1}/{len(self.files)}] {filename} | " f"Shift: {self.current_shift}px\nFiles: {schema['number_of_files']} | " f"Cols: {schema['columns_per_file']}\n" "Enter: Save and next | s: flag error and skip | n: +50 | N: +100 | t: -50 | " "1: use single column" ), fg="black", ) def on_shift(self, amount: int) -> None: if self.is_processing: return self.current_shift += amount print(f"Applying shift: {self.current_shift}") self.trigger_processing(self.files[self.index], self.current_shift) def on_next(self, _event: object) -> None: if self.is_processing or self.current_result is None: return pdf_path = self.files[self.index] try: save_results(self.current_result, pdf_path, self.output_dir) clear_copy_error(self.workspace, pdf_path) except Exception as exc: self._mark_error(pdf_path, f"Échec de l’enregistrement : {exc}") messagebox.showerror("Enregistrement impossible", str(exc), parent=self.root) return self._advance() def _mark_error(self, pdf_path: Path, reason: str) -> None: mark_copy_error(self.workspace, pdf_path, reason) self.had_errors = True print(f"[Copie signalée] {pdf_path.name}: {reason}") def on_skip(self, _event=None) -> None: if self.is_processing or self.index >= len(self.files): return self._mark_error(self.files[self.index], "Problème repéré pendant la découpe des marges") self._advance() def on_close(self) -> None: self.stop_prefetch.set() self.root.destroy() def _advance(self) -> None: self.index += 1 self.current_shift = 0 self.current_max_per_file = self.default_max_per_file self.load_current_image() def _selected_files( workspace: EvaluationWorkspace, target: Path, ) -> list[Path]: workspace.require_directories("Copies") if target.is_file(): if target.suffix.casefold() != ".pdf": raise CliError(f"Target is not a PDF: {target}", ExitCode.INVALID_ARGUMENTS) return [target] return sorted( ( path for path in workspace.copies_dir.glob("*.pdf") if "nonc" not in path.name.casefold() ), key=lambda path: path.name.casefold(), ) def run( workspace: EvaluationWorkspace, target: Path, *, fullpage: bool = False, marked: bool = False, ) -> ExitCode: files = marked_copy_paths(workspace) if marked else _selected_files(workspace, target) if not files: print("No PDF files found.") return ExitCode.SUCCESS workspace.cutleft_dir.mkdir(parents=True, exist_ok=True) _get_pdf_pages_cached.cache_clear() reviewer = ImageReviewer( files, workspace.cutleft_dir, default_max_per_file=1 if fullpage else 5, ) if not reviewer.completed: return ExitCode.INTERRUPTED return ExitCode.PARTIAL if reviewer.had_errors else ExitCode.SUCCESS def build_parser() -> argparse.ArgumentParser: parser = target_parser("Interactively crop the label margin from PDF copies") parser.add_argument("--marked", action="store_true", help="Review flagged copies and clear each flag after saving with Enter") parser.add_argument( "--fullpage", action="store_true", help="Use each complete page instead of cropping the label margin", ) return parser def main(argv: Sequence[str] | None = None) -> int: parser = build_parser() def handle(args: argparse.Namespace) -> ExitCode: workspace, target = workspace_from_target(args) return run(workspace, target, fullpage=args.fullpage, marked=args.marked) return execute(parser, argv, handle) if __name__ == "__main__": raise SystemExit(main())