Files
Copies/copienator/commands/cutleft.py
T
2026-09-08 16:41:04 +02:00

387 lines
14 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
from __future__ import annotations
import argparse
import threading
import tkinter as tk
from tkinter import messagebox
from collections.abc import Sequence
from functools import lru_cache
from pathlib import Path
from queue import Empty, Queue
from threading import Thread
from pdf2image import convert_from_path
from PIL import Image, ImageTk
from copienator import (
CliError,
EvaluationWorkspace,
ExitCode,
atomic_write_json,
execute,
target_parser,
workspace_from_target,
)
from copienator.filesystem import staged_files
from copienator.copy_errors import clear_copy_error, mark_copy_error, marked_copy_paths
DELIMITER_WIDTH = 5
DELIMITER_COLOR = (0, 0, 0)
OUTPUT_SIZE = (1800, 1000)
pdf_cache_lock = threading.Lock()
def distribute_pages(total_pages: int, max_per_file: int = 5) -> list[int]:
"""Distribute pages into balanced chunks no larger than max_per_file."""
if total_pages == 0:
return []
number_of_files = (total_pages + max_per_file - 1) // max_per_file
base_count, remainder = divmod(total_pages, number_of_files)
return [
base_count + (1 if index < remainder else 0)
for index in range(number_of_files)
]
def stitch_images(image_list: list[Image.Image]) -> Image.Image | None:
if not image_list:
return None
total_width = sum(image.width for image in image_list)
total_width += (len(image_list) - 1) * DELIMITER_WIDTH
max_height = max(image.height for image in image_list)
combined = Image.new("RGB", (total_width, max_height), color="white")
x_offset = 0
for index, image in enumerate(image_list):
combined.paste(image, (x_offset, 0))
x_offset += image.width
if index < len(image_list) - 1:
delimiter = Image.new(
"RGB", (DELIMITER_WIDTH, max_height), color=DELIMITER_COLOR
)
combined.paste(delimiter, (x_offset, 0))
x_offset += DELIMITER_WIDTH
return combined
@lru_cache(maxsize=3)
def _get_pdf_pages_cached(pdf_path: Path) -> list[Image.Image]:
return convert_from_path(pdf_path)
def get_pdf_pages(pdf_path: Path) -> list[Image.Image]:
"""Thread-safe wrapper around the small PDF conversion cache."""
with pdf_cache_lock:
return _get_pdf_pages_cached(pdf_path)
def process_single_pdf(
pdf_path: Path,
shift_offset: int = 0,
max_per_file: int = 5,
) -> tuple[Image.Image, list[Image.Image], dict[str, object]] | None:
"""Convert one PDF into a preview, full-resolution splits and metadata."""
try:
cropped_images = []
for image in get_pdf_pages(pdf_path):
width, height = image.size
if max_per_file == 1:
left, right = 0, width
else:
left = max(0, 100 + shift_offset)
right = min(width, width // 3 + 100 + shift_offset)
if right > left:
cropped_images.append(image.crop((left, 0, right, height)))
if not cropped_images:
return None
distribution = distribute_pages(len(cropped_images), max_per_file)
split_images = []
current_index = 0
for count in distribution:
stitched = stitch_images(cropped_images[current_index : current_index + count])
if stitched is not None:
split_images.append(stitched)
current_index += count
full_stitch = stitch_images(cropped_images)
if full_stitch is None:
return None
preview = full_stitch.resize(OUTPUT_SIZE, Image.Resampling.BILINEAR)
schema: dict[str, object] = {
"original_filename": pdf_path.name,
"total_pages": len(cropped_images),
"number_of_files": len(split_images),
"columns_per_file": distribution,
}
return preview, split_images, schema
except Exception as exc: # noqa: BLE001 - interactive item failure
print(f"Error processing {pdf_path.name}: {exc}")
return None
def _previous_cutleft_outputs(output_dir: Path, base_name: str) -> set[str]:
if not output_dir.is_dir():
return set()
result = {f"{base_name}_schema.json"}
for path in output_dir.glob(f"{base_name}_*.jpg"):
suffix = path.stem.removeprefix(f"{base_name}_")
if suffix.isdigit():
result.add(path.name)
return result
def save_results(
result: tuple[Image.Image, list[Image.Image], dict[str, object]],
pdf_path: Path,
output_dir: Path,
) -> None:
"""Atomically replace every Cutleft output associated with one copy."""
_, splits, schema = result
base_name = pdf_path.stem
previous = _previous_cutleft_outputs(output_dir, base_name)
with staged_files(output_dir, remove=previous) as staging:
for index, image in enumerate(splits, start=1):
filename = f"{base_name}_{index:02d}.jpg"
image.save(staging / filename, "JPEG", quality=95)
atomic_write_json(staging / f"{base_name}_schema.json", schema)
for index in range(1, len(splits) + 1):
print(f"Saved: {base_name}_{index:02d}.jpg")
print(f"Saved schema: {base_name}_schema.json")
class ImageReviewer:
def __init__(
self,
files: list[Path],
output_dir: Path,
default_max_per_file: int = 5,
) -> None:
self.files = files
self.output_dir = output_dir
self.workspace = EvaluationWorkspace(output_dir.parent)
self.completed = False
self.had_errors = False
self.stop_prefetch = threading.Event()
self.current_result = None
self.index = 0
self.current_shift = 0
self.default_max_per_file = default_max_per_file
self.current_max_per_file = default_max_per_file
self.current_preview: Image.Image | None = None
self.is_processing = False
self.manual_queue: Queue[
tuple[Image.Image, list[Image.Image], dict[str, object]] | None
] = Queue()
self.root = tk.Tk()
self.root.title("PDF Cropper")
self.root.geometry("+100+100")
self.label_img = tk.Label(self.root)
self.label_img.pack()
self.label_info = tk.Label(self.root, text="", font=("Arial", 12, "bold"))
self.label_info.pack(pady=5)
self.root.bind("<Return>", self.on_next)
self.root.bind("s", self.on_skip)
self.root.protocol("WM_DELETE_WINDOW", self.on_close)
self.root.bind("n", lambda _event: self.on_shift(50))
self.root.bind("N", lambda _event: self.on_shift(100))
self.root.bind("t", lambda _event: self.on_shift(-50))
self.root.bind("1", lambda _event: self.on_set_max_pages(1))
Thread(target=self.prefetch_worker, daemon=True).start()
self.load_current_image()
self.root.lift()
self.root.focus_force()
self.root.mainloop()
def on_set_max_pages(self, count: int) -> None:
if self.is_processing:
return
self.current_max_per_file = count
print(f"Setting max pages per file: {count}")
self.trigger_processing(self.files[self.index], self.current_shift)
def prefetch_worker(self) -> None:
processed_index = -1
while not self.stop_prefetch.is_set():
target = self.index + 1
if target < len(self.files) and target != processed_index:
try:
get_pdf_pages(self.files[target])
except Exception:
pass # The foreground review reports and flags conversion errors.
processed_index = target
self.stop_prefetch.wait(0.05)
def load_current_image(self) -> None:
if self.index >= len(self.files):
print("All files processed.")
self.completed = True
self.on_close()
return
self.is_processing = False
self.current_shift = 0
self.current_result = None
self.trigger_processing(self.files[self.index], self.current_shift)
def trigger_processing(self, pdf_path: Path, shift: int) -> None:
self.is_processing = True
self.label_info.configure(
text=f"Processing {pdf_path.name} (Shift {shift})... Please wait.",
fg="red",
)
def worker() -> None:
self.manual_queue.put(
process_single_pdf(pdf_path, shift, self.current_max_per_file)
)
Thread(target=worker, daemon=True).start()
self.check_manual_queue(pdf_path)
def check_manual_queue(self, pdf_path: Path) -> None:
try:
result = self.manual_queue.get_nowait()
self.is_processing = False
if result is None:
print(f"Failed to process {pdf_path.name}, skipping.")
self._mark_error(pdf_path, "Échec de la conversion pour la découpe des marges")
self._advance()
else:
self.handle_processing_result(result, pdf_path)
except Empty:
self.root.after(100, lambda: self.check_manual_queue(pdf_path))
def handle_processing_result(
self,
result: tuple[Image.Image, list[Image.Image], dict[str, object]],
pdf_path: Path,
) -> None:
self.current_preview = result[0]
self.current_result = result
self.update_display(pdf_path.name, result[2])
def update_display(self, filename: str, schema: dict[str, object]) -> None:
if self.current_preview is None:
return
tk_image = ImageTk.PhotoImage(self.current_preview)
self.label_img.configure(image=tk_image)
self.label_img.image = tk_image
self.label_info.configure(
text=(
f"[{self.index + 1}/{len(self.files)}] {filename} | "
f"Shift: {self.current_shift}px\nFiles: {schema['number_of_files']} | "
f"Cols: {schema['columns_per_file']}\n"
"Enter: Save and next | s: flag error and skip | n: +50 | N: +100 | t: -50 | "
"1: use single column"
),
fg="black",
)
def on_shift(self, amount: int) -> None:
if self.is_processing:
return
self.current_shift += amount
print(f"Applying shift: {self.current_shift}")
self.trigger_processing(self.files[self.index], self.current_shift)
def on_next(self, _event: object) -> None:
if self.is_processing or self.current_result is None:
return
pdf_path = self.files[self.index]
try:
save_results(self.current_result, pdf_path, self.output_dir)
clear_copy_error(self.workspace, pdf_path)
except Exception as exc:
self._mark_error(pdf_path, f"Échec de lenregistrement : {exc}")
messagebox.showerror("Enregistrement impossible", str(exc), parent=self.root)
return
self._advance()
def _mark_error(self, pdf_path: Path, reason: str) -> None:
mark_copy_error(self.workspace, pdf_path, reason)
self.had_errors = True
print(f"[Copie signalée] {pdf_path.name}: {reason}")
def on_skip(self, _event=None) -> None:
if self.is_processing or self.index >= len(self.files):
return
self._mark_error(self.files[self.index], "Problème repéré pendant la découpe des marges")
self._advance()
def on_close(self) -> None:
self.stop_prefetch.set()
self.root.destroy()
def _advance(self) -> None:
self.index += 1
self.current_shift = 0
self.current_max_per_file = self.default_max_per_file
self.load_current_image()
def _selected_files(
workspace: EvaluationWorkspace,
target: Path,
) -> list[Path]:
workspace.require_directories("Copies")
if target.is_file():
if target.suffix.casefold() != ".pdf":
raise CliError(f"Target is not a PDF: {target}", ExitCode.INVALID_ARGUMENTS)
return [target]
return sorted(
(
path
for path in workspace.copies_dir.glob("*.pdf")
if "nonc" not in path.name.casefold()
),
key=lambda path: path.name.casefold(),
)
def run(
workspace: EvaluationWorkspace,
target: Path,
*,
fullpage: bool = False,
marked: bool = False,
) -> ExitCode:
files = marked_copy_paths(workspace) if marked else _selected_files(workspace, target)
if not files:
print("No PDF files found.")
return ExitCode.SUCCESS
workspace.cutleft_dir.mkdir(parents=True, exist_ok=True)
_get_pdf_pages_cached.cache_clear()
reviewer = ImageReviewer(
files,
workspace.cutleft_dir,
default_max_per_file=1 if fullpage else 5,
)
if not reviewer.completed:
return ExitCode.INTERRUPTED
return ExitCode.PARTIAL if reviewer.had_errors else ExitCode.SUCCESS
def build_parser() -> argparse.ArgumentParser:
parser = target_parser("Interactively crop the label margin from PDF copies")
parser.add_argument("--marked", action="store_true", help="Review flagged copies and clear each flag after saving with Enter")
parser.add_argument(
"--fullpage",
action="store_true",
help="Use each complete page instead of cropping the label margin",
)
return parser
def main(argv: Sequence[str] | None = None) -> int:
parser = build_parser()
def handle(args: argparse.Namespace) -> ExitCode:
workspace, target = workspace_from_target(args)
return run(workspace, target, fullpage=args.fullpage, marked=args.marked)
return execute(parser, argv, handle)
if __name__ == "__main__":
raise SystemExit(main())