389 lines
14 KiB
Python
389 lines
14 KiB
Python
from __future__ import annotations
|
||
|
||
import argparse
|
||
import threading
|
||
import tkinter as tk
|
||
from tkinter import messagebox
|
||
from collections.abc import Sequence
|
||
from functools import lru_cache
|
||
from pathlib import Path
|
||
from queue import Empty, Queue
|
||
from threading import Thread
|
||
|
||
from pdf2image import convert_from_path
|
||
from PIL import Image, ImageTk
|
||
|
||
from copienator import (
|
||
CliError,
|
||
EvaluationWorkspace,
|
||
ExitCode,
|
||
atomic_write_json,
|
||
execute,
|
||
target_parser,
|
||
workspace_from_target,
|
||
)
|
||
from copienator.filesystem import staged_files
|
||
from copienator.copy_errors import clear_copy_error, mark_copy_error, marked_copy_paths
|
||
|
||
DELIMITER_WIDTH = 5
|
||
DELIMITER_COLOR = (0, 0, 0)
|
||
OUTPUT_SIZE = (1800, 1000)
|
||
pdf_cache_lock = threading.Lock()
|
||
|
||
|
||
def distribute_pages(total_pages: int, max_per_file: int = 5) -> list[int]:
|
||
"""Distribute pages into balanced chunks no larger than max_per_file."""
|
||
if total_pages == 0:
|
||
return []
|
||
number_of_files = (total_pages + max_per_file - 1) // max_per_file
|
||
base_count, remainder = divmod(total_pages, number_of_files)
|
||
return [
|
||
base_count + (1 if index < remainder else 0)
|
||
for index in range(number_of_files)
|
||
]
|
||
|
||
|
||
def stitch_images(image_list: list[Image.Image]) -> Image.Image | None:
|
||
if not image_list:
|
||
return None
|
||
total_width = sum(image.width for image in image_list)
|
||
total_width += (len(image_list) - 1) * DELIMITER_WIDTH
|
||
max_height = max(image.height for image in image_list)
|
||
combined = Image.new("RGB", (total_width, max_height), color="white")
|
||
x_offset = 0
|
||
for index, image in enumerate(image_list):
|
||
combined.paste(image, (x_offset, 0))
|
||
x_offset += image.width
|
||
if index < len(image_list) - 1:
|
||
delimiter = Image.new(
|
||
"RGB", (DELIMITER_WIDTH, max_height), color=DELIMITER_COLOR
|
||
)
|
||
combined.paste(delimiter, (x_offset, 0))
|
||
x_offset += DELIMITER_WIDTH
|
||
return combined
|
||
|
||
|
||
@lru_cache(maxsize=3)
|
||
def _get_pdf_pages_cached(pdf_path: Path) -> list[Image.Image]:
|
||
# Label coordinates use the full, displayed MediaBox, including on PDFs
|
||
# previously cropped by crop-margins. Keep this in sync with split-answers.
|
||
return convert_from_path(pdf_path, use_cropbox=False)
|
||
|
||
|
||
def get_pdf_pages(pdf_path: Path) -> list[Image.Image]:
|
||
"""Thread-safe wrapper around the small PDF conversion cache."""
|
||
with pdf_cache_lock:
|
||
return _get_pdf_pages_cached(pdf_path)
|
||
|
||
|
||
def process_single_pdf(
|
||
pdf_path: Path,
|
||
shift_offset: int = 0,
|
||
max_per_file: int = 5,
|
||
) -> tuple[Image.Image, list[Image.Image], dict[str, object]] | None:
|
||
"""Convert one PDF into a preview, full-resolution splits and metadata."""
|
||
try:
|
||
cropped_images = []
|
||
for image in get_pdf_pages(pdf_path):
|
||
width, height = image.size
|
||
if max_per_file == 1:
|
||
left, right = 0, width
|
||
else:
|
||
left = max(0, 100 + shift_offset)
|
||
right = min(width, width // 3 + 100 + shift_offset)
|
||
if right > left:
|
||
cropped_images.append(image.crop((left, 0, right, height)))
|
||
if not cropped_images:
|
||
return None
|
||
|
||
distribution = distribute_pages(len(cropped_images), max_per_file)
|
||
split_images = []
|
||
current_index = 0
|
||
for count in distribution:
|
||
stitched = stitch_images(cropped_images[current_index : current_index + count])
|
||
if stitched is not None:
|
||
split_images.append(stitched)
|
||
current_index += count
|
||
full_stitch = stitch_images(cropped_images)
|
||
if full_stitch is None:
|
||
return None
|
||
preview = full_stitch.resize(OUTPUT_SIZE, Image.Resampling.BILINEAR)
|
||
schema: dict[str, object] = {
|
||
"original_filename": pdf_path.name,
|
||
"total_pages": len(cropped_images),
|
||
"number_of_files": len(split_images),
|
||
"columns_per_file": distribution,
|
||
}
|
||
return preview, split_images, schema
|
||
except Exception as exc: # noqa: BLE001 - interactive item failure
|
||
print(f"Error processing {pdf_path.name}: {exc}")
|
||
return None
|
||
|
||
|
||
def _previous_cutleft_outputs(output_dir: Path, base_name: str) -> set[str]:
|
||
if not output_dir.is_dir():
|
||
return set()
|
||
result = {f"{base_name}_schema.json"}
|
||
for path in output_dir.glob(f"{base_name}_*.jpg"):
|
||
suffix = path.stem.removeprefix(f"{base_name}_")
|
||
if suffix.isdigit():
|
||
result.add(path.name)
|
||
return result
|
||
|
||
|
||
def save_results(
|
||
result: tuple[Image.Image, list[Image.Image], dict[str, object]],
|
||
pdf_path: Path,
|
||
output_dir: Path,
|
||
) -> None:
|
||
"""Atomically replace every Cutleft output associated with one copy."""
|
||
_, splits, schema = result
|
||
base_name = pdf_path.stem
|
||
previous = _previous_cutleft_outputs(output_dir, base_name)
|
||
with staged_files(output_dir, remove=previous) as staging:
|
||
for index, image in enumerate(splits, start=1):
|
||
filename = f"{base_name}_{index:02d}.jpg"
|
||
image.save(staging / filename, "JPEG", quality=95)
|
||
atomic_write_json(staging / f"{base_name}_schema.json", schema)
|
||
for index in range(1, len(splits) + 1):
|
||
print(f"Saved: {base_name}_{index:02d}.jpg")
|
||
print(f"Saved schema: {base_name}_schema.json")
|
||
|
||
|
||
class ImageReviewer:
|
||
def __init__(
|
||
self,
|
||
files: list[Path],
|
||
output_dir: Path,
|
||
default_max_per_file: int = 5,
|
||
) -> None:
|
||
self.files = files
|
||
self.output_dir = output_dir
|
||
self.workspace = EvaluationWorkspace(output_dir.parent)
|
||
self.completed = False
|
||
self.had_errors = False
|
||
self.stop_prefetch = threading.Event()
|
||
self.current_result = None
|
||
self.index = 0
|
||
self.current_shift = 0
|
||
self.default_max_per_file = default_max_per_file
|
||
self.current_max_per_file = default_max_per_file
|
||
self.current_preview: Image.Image | None = None
|
||
self.is_processing = False
|
||
self.manual_queue: Queue[
|
||
tuple[Image.Image, list[Image.Image], dict[str, object]] | None
|
||
] = Queue()
|
||
|
||
self.root = tk.Tk()
|
||
self.root.title("PDF Cropper")
|
||
self.root.geometry("+100+100")
|
||
self.label_img = tk.Label(self.root)
|
||
self.label_img.pack()
|
||
self.label_info = tk.Label(self.root, text="", font=("Arial", 12, "bold"))
|
||
self.label_info.pack(pady=5)
|
||
self.root.bind("<Return>", self.on_next)
|
||
self.root.bind("s", self.on_skip)
|
||
self.root.protocol("WM_DELETE_WINDOW", self.on_close)
|
||
self.root.bind("n", lambda _event: self.on_shift(50))
|
||
self.root.bind("N", lambda _event: self.on_shift(100))
|
||
self.root.bind("t", lambda _event: self.on_shift(-50))
|
||
self.root.bind("1", lambda _event: self.on_set_max_pages(1))
|
||
|
||
Thread(target=self.prefetch_worker, daemon=True).start()
|
||
self.load_current_image()
|
||
self.root.lift()
|
||
self.root.focus_force()
|
||
self.root.mainloop()
|
||
|
||
def on_set_max_pages(self, count: int) -> None:
|
||
if self.is_processing:
|
||
return
|
||
self.current_max_per_file = count
|
||
print(f"Setting max pages per file: {count}")
|
||
self.trigger_processing(self.files[self.index], self.current_shift)
|
||
|
||
def prefetch_worker(self) -> None:
|
||
processed_index = -1
|
||
while not self.stop_prefetch.is_set():
|
||
target = self.index + 1
|
||
if target < len(self.files) and target != processed_index:
|
||
try:
|
||
get_pdf_pages(self.files[target])
|
||
except Exception:
|
||
pass # The foreground review reports and flags conversion errors.
|
||
processed_index = target
|
||
self.stop_prefetch.wait(0.05)
|
||
|
||
def load_current_image(self) -> None:
|
||
if self.index >= len(self.files):
|
||
print("All files processed.")
|
||
self.completed = True
|
||
self.on_close()
|
||
return
|
||
self.is_processing = False
|
||
self.current_shift = 0
|
||
self.current_result = None
|
||
self.trigger_processing(self.files[self.index], self.current_shift)
|
||
|
||
def trigger_processing(self, pdf_path: Path, shift: int) -> None:
|
||
self.is_processing = True
|
||
self.label_info.configure(
|
||
text=f"Processing {pdf_path.name} (Shift {shift})... Please wait.",
|
||
fg="red",
|
||
)
|
||
|
||
def worker() -> None:
|
||
self.manual_queue.put(
|
||
process_single_pdf(pdf_path, shift, self.current_max_per_file)
|
||
)
|
||
|
||
Thread(target=worker, daemon=True).start()
|
||
self.check_manual_queue(pdf_path)
|
||
|
||
def check_manual_queue(self, pdf_path: Path) -> None:
|
||
try:
|
||
result = self.manual_queue.get_nowait()
|
||
self.is_processing = False
|
||
if result is None:
|
||
print(f"Failed to process {pdf_path.name}, skipping.")
|
||
self._mark_error(pdf_path, "Échec de la conversion pour la découpe des marges")
|
||
self._advance()
|
||
else:
|
||
self.handle_processing_result(result, pdf_path)
|
||
except Empty:
|
||
self.root.after(100, lambda: self.check_manual_queue(pdf_path))
|
||
|
||
def handle_processing_result(
|
||
self,
|
||
result: tuple[Image.Image, list[Image.Image], dict[str, object]],
|
||
pdf_path: Path,
|
||
) -> None:
|
||
self.current_preview = result[0]
|
||
self.current_result = result
|
||
self.update_display(pdf_path.name, result[2])
|
||
|
||
def update_display(self, filename: str, schema: dict[str, object]) -> None:
|
||
if self.current_preview is None:
|
||
return
|
||
tk_image = ImageTk.PhotoImage(self.current_preview)
|
||
self.label_img.configure(image=tk_image)
|
||
self.label_img.image = tk_image
|
||
self.label_info.configure(
|
||
text=(
|
||
f"[{self.index + 1}/{len(self.files)}] {filename} | "
|
||
f"Shift: {self.current_shift}px\nFiles: {schema['number_of_files']} | "
|
||
f"Cols: {schema['columns_per_file']}\n"
|
||
"Enter: Save and next | s: flag error and skip | n: +50 | N: +100 | t: -50 | "
|
||
"1: use single column"
|
||
),
|
||
fg="black",
|
||
)
|
||
|
||
def on_shift(self, amount: int) -> None:
|
||
if self.is_processing:
|
||
return
|
||
self.current_shift += amount
|
||
print(f"Applying shift: {self.current_shift}")
|
||
self.trigger_processing(self.files[self.index], self.current_shift)
|
||
|
||
def on_next(self, _event: object) -> None:
|
||
if self.is_processing or self.current_result is None:
|
||
return
|
||
pdf_path = self.files[self.index]
|
||
try:
|
||
save_results(self.current_result, pdf_path, self.output_dir)
|
||
clear_copy_error(self.workspace, pdf_path)
|
||
except Exception as exc:
|
||
self._mark_error(pdf_path, f"Échec de l’enregistrement : {exc}")
|
||
messagebox.showerror("Enregistrement impossible", str(exc), parent=self.root)
|
||
return
|
||
self._advance()
|
||
|
||
def _mark_error(self, pdf_path: Path, reason: str) -> None:
|
||
mark_copy_error(self.workspace, pdf_path, reason)
|
||
self.had_errors = True
|
||
print(f"[Copie signalée] {pdf_path.name}: {reason}")
|
||
|
||
def on_skip(self, _event=None) -> None:
|
||
if self.is_processing or self.index >= len(self.files):
|
||
return
|
||
self._mark_error(self.files[self.index], "Problème repéré pendant la découpe des marges")
|
||
self._advance()
|
||
|
||
def on_close(self) -> None:
|
||
self.stop_prefetch.set()
|
||
self.root.destroy()
|
||
|
||
def _advance(self) -> None:
|
||
self.index += 1
|
||
self.current_shift = 0
|
||
self.current_max_per_file = self.default_max_per_file
|
||
self.load_current_image()
|
||
|
||
|
||
def _selected_files(
|
||
workspace: EvaluationWorkspace,
|
||
target: Path,
|
||
) -> list[Path]:
|
||
workspace.require_directories("Copies")
|
||
if target.is_file():
|
||
if target.suffix.casefold() != ".pdf":
|
||
raise CliError(f"Target is not a PDF: {target}", ExitCode.INVALID_ARGUMENTS)
|
||
return [target]
|
||
return sorted(
|
||
(
|
||
path
|
||
for path in workspace.copies_dir.glob("*.pdf")
|
||
if "nonc" not in path.name.casefold()
|
||
),
|
||
key=lambda path: path.name.casefold(),
|
||
)
|
||
|
||
|
||
def run(
|
||
workspace: EvaluationWorkspace,
|
||
target: Path,
|
||
*,
|
||
fullpage: bool = False,
|
||
marked: bool = False,
|
||
) -> ExitCode:
|
||
files = marked_copy_paths(workspace) if marked else _selected_files(workspace, target)
|
||
if not files:
|
||
print("No PDF files found.")
|
||
return ExitCode.SUCCESS
|
||
workspace.cutleft_dir.mkdir(parents=True, exist_ok=True)
|
||
_get_pdf_pages_cached.cache_clear()
|
||
reviewer = ImageReviewer(
|
||
files,
|
||
workspace.cutleft_dir,
|
||
default_max_per_file=1 if fullpage else 5,
|
||
)
|
||
if not reviewer.completed:
|
||
return ExitCode.INTERRUPTED
|
||
return ExitCode.PARTIAL if reviewer.had_errors else ExitCode.SUCCESS
|
||
|
||
|
||
def build_parser() -> argparse.ArgumentParser:
|
||
parser = target_parser("Interactively crop the label margin from PDF copies")
|
||
parser.add_argument("--marked", action="store_true", help="Review flagged copies and clear each flag after saving with Enter")
|
||
parser.add_argument(
|
||
"--fullpage",
|
||
action="store_true",
|
||
help="Use each complete page instead of cropping the label margin",
|
||
)
|
||
return parser
|
||
|
||
|
||
def main(argv: Sequence[str] | None = None) -> int:
|
||
parser = build_parser()
|
||
|
||
def handle(args: argparse.Namespace) -> ExitCode:
|
||
workspace, target = workspace_from_target(args)
|
||
return run(workspace, target, fullpage=args.fullpage, marked=args.marked)
|
||
|
||
return execute(parser, argv, handle)
|
||
|
||
|
||
if __name__ == "__main__":
|
||
raise SystemExit(main())
|