"""Optionally replace split-answer PDFs with large bottom crops.""" from __future__ import annotations import argparse import hashlib import json import multiprocessing import shutil import signal import tempfile from collections.abc import Sequence from pathlib import Path import cv2 from copienator.cli import CliError, ExitCode, execute, target_parser, workspace_from_target from copienator.crop_exercise_bottoms import _full_page_height, process_exercise_pdf from copienator.workspace import EvaluationWorkspace def selected_files(workspace: EvaluationWorkspace, target: Path) -> list[Path]: workspace.require_directories("Copies") resolved = target.resolve() if resolved not in {workspace.root.resolve(), workspace.copies_dir.resolve()}: raise CliError( "La cible doit être l’évaluation ou son dossier Copies.", ExitCode.INVALID_ARGUMENTS, ) files = sorted( workspace.copies_dir.glob("Copie*/*.pdf"), key=lambda path: (path.parent.name.casefold(), path.name.casefold()), ) for source in files: if source.is_symlink(): raise CliError(f"Lien symbolique non pris en charge : {source}") return files def _initialize_worker() -> None: cv2.setNumThreads(1) signal.signal(signal.SIGINT, signal.SIG_IGN) def _process_file(job: tuple[Path, Path, float]) -> list[dict]: source, destination, full_height = job records = process_exercise_pdf( source, destination, None, full_height, dpi=200, padding_mm=6 ) changed = sum(record["status"] == "cropped" for record in records) if changed: print( f"{source.parent.name}/{source.name} : {changed}/{len(records)} page(s) rognée(s)", flush=True, ) return records def process_files( workspace: EvaluationWorkspace, files: list[Path], staging: Path, workers: int, ) -> list[dict]: heights: dict[str, float] = {} for copy_name in sorted({source.parent.name for source in files}): copy_pdf = workspace.copies_dir / f"{copy_name}.pdf" if not copy_pdf.is_file(): raise CliError(f"PDF source introuvable : {copy_pdf}") heights[copy_name] = _full_page_height(copy_pdf) jobs = [ ( source, staging / source.relative_to(workspace.copies_dir), heights[source.parent.name], ) for source in files ] count = min(workers, len(jobs)) print( f"Analyse de {len(files)} PDF de réponses avec {count} traitement(s) en parallèle.", flush=True, ) if count == 1: previous_threads = cv2.getNumThreads() cv2.setNumThreads(1) try: batches = [_process_file(job) for job in jobs] finally: cv2.setNumThreads(previous_threads) else: with multiprocessing.get_context("spawn").Pool( count, _initialize_worker ) as pool: batches = list(pool.imap_unordered(_process_file, jobs)) order = {source.as_posix(): index for index, source in enumerate(files)} return sorted( (record for batch in batches for record in batch), key=lambda record: (order[record["file"]], record["page"]), ) def _publish( workspace: EvaluationWorkspace, changed_files: list[Path], staging: Path, records: list[dict], ) -> Path: workspace.runs_dir.mkdir(parents=True, exist_ok=True) run_dir = Path( tempfile.mkdtemp(prefix="crop-exercise-bottoms-", dir=workspace.runs_dir) ) backup_root = run_dir / "Copies" replaced: list[Path] = [] try: for source in changed_files: backup = backup_root / source.relative_to(workspace.copies_dir) backup.parent.mkdir(parents=True, exist_ok=True) shutil.copy2(source, backup) (run_dir / "report.json").write_text( json.dumps(records, ensure_ascii=False, indent=2) + "\n", encoding="utf-8" ) for source in changed_files: prepared = staging / source.relative_to(workspace.copies_dir) prepared.replace(source) replaced.append(source) except BaseException: for source in replaced: backup = backup_root / source.relative_to(workspace.copies_dir) if backup.is_file(): shutil.copy2(backup, source) shutil.rmtree(run_dir, ignore_errors=True) raise return backup_root def crop_statistics(records: list[dict]) -> tuple[int, float]: """Return cropped exercise count and mean removed percentage per exercise.""" totals: dict[str, list[float]] = {} cropped_files: set[str] = set() for record in records: total_height, removed_height = totals.setdefault(record["file"], [0.0, 0.0]) totals[record["file"]] = [ total_height + float(record["height_lines"]), removed_height + float(record["bottom_removed_lines"]), ] if record["status"] == "cropped": cropped_files.add(record["file"]) percentages = [ min(100.0, totals[file_name][1] / totals[file_name][0] * 100) for file_name in cropped_files if totals[file_name][0] > 0 ] mean_percentage = sum(percentages) / len(percentages) if percentages else 0.0 return len(cropped_files), mean_percentage def run(workspace: EvaluationWorkspace, target: Path, *, workers: int = 5) -> ExitCode: if workers < 1: raise CliError( "Le nombre de traitements parallèles doit être positif.", ExitCode.INVALID_ARGUMENTS, ) files = selected_files(workspace, target) if not files: raise CliError( "Aucun PDF de réponse trouvé dans Copies/CopieXX/.", ExitCode.INVALID_ARGUMENTS, ) with tempfile.TemporaryDirectory( prefix=".crop-exercise-bottoms-", dir=workspace.root ) as directory: staging = Path(directory) records = process_files(workspace, files, staging, workers) changed_names = { record["file"] for record in records if record["status"] == "cropped" } changed_files = [source for source in files if source.as_posix() in changed_names] digests = {record["file"]: record["source_sha256"] for record in records} for source in files: if hashlib.sha256(source.read_bytes()).hexdigest() != digests[source.as_posix()]: raise CliError( f"{source} a changé pendant l’analyse. Aucun PDF remplacé." ) cropped_exercises, mean_percentage = crop_statistics(records) if not changed_files: print("Terminé : aucun exercice ne remplit les critères de rognage.", flush=True) print( "Rognage moyen des exercices modifiés : 0.0 %.", flush=True ) return ExitCode.SUCCESS backup = _publish(workspace, changed_files, staging, records) cropped_pages = sum(record["status"] == "cropped" for record in records) print(f"Sauvegarde des PDF non rognés : {backup}", flush=True) print( f"Terminé : {cropped_exercises} exercice(s) rogné(s), soit " f"{cropped_pages} page(s) dans {len(changed_files)} PDF remplacé(s).", flush=True, ) print( f"Rognage moyen des exercices modifiés : {mean_percentage:.1f} %.", flush=True, ) return ExitCode.SUCCESS def main(argv: Sequence[str] | None = None) -> int: parser = target_parser( "Rogner les grands espaces vides au bas des réponses déjà découpées" ) parser.add_argument( "--workers", type=int, default=5, help="Nombre de PDF traités en parallèle (défaut : 5)", ) def handle(arguments: argparse.Namespace) -> ExitCode: workspace, target = workspace_from_target(arguments) return run(workspace, target, workers=arguments.workers) return execute(parser, argv, handle) if __name__ == "__main__": raise SystemExit(main())