227 lines
7.9 KiB
Python
227 lines
7.9 KiB
Python
"""Optionally replace split-answer PDFs with large bottom crops."""
|
||
|
||
from __future__ import annotations
|
||
|
||
import argparse
|
||
import hashlib
|
||
import json
|
||
import multiprocessing
|
||
import shutil
|
||
import signal
|
||
import tempfile
|
||
from collections.abc import Sequence
|
||
from pathlib import Path
|
||
|
||
import cv2
|
||
|
||
from copienator.cli import CliError, ExitCode, execute, target_parser, workspace_from_target
|
||
from copienator.crop_exercise_bottoms import _full_page_height, process_exercise_pdf
|
||
from copienator.workspace import EvaluationWorkspace
|
||
|
||
|
||
def selected_files(workspace: EvaluationWorkspace, target: Path) -> list[Path]:
|
||
workspace.require_directories("Copies")
|
||
resolved = target.resolve()
|
||
if resolved not in {workspace.root.resolve(), workspace.copies_dir.resolve()}:
|
||
raise CliError(
|
||
"La cible doit être l’évaluation ou son dossier Copies.",
|
||
ExitCode.INVALID_ARGUMENTS,
|
||
)
|
||
files = sorted(
|
||
workspace.copies_dir.glob("Copie*/*.pdf"),
|
||
key=lambda path: (path.parent.name.casefold(), path.name.casefold()),
|
||
)
|
||
for source in files:
|
||
if source.is_symlink():
|
||
raise CliError(f"Lien symbolique non pris en charge : {source}")
|
||
return files
|
||
|
||
|
||
def _initialize_worker() -> None:
|
||
cv2.setNumThreads(1)
|
||
signal.signal(signal.SIGINT, signal.SIG_IGN)
|
||
|
||
|
||
def _process_file(job: tuple[Path, Path, float]) -> list[dict]:
|
||
source, destination, full_height = job
|
||
records = process_exercise_pdf(
|
||
source, destination, None, full_height, dpi=200, padding_mm=6
|
||
)
|
||
changed = sum(record["status"] == "cropped" for record in records)
|
||
if changed:
|
||
print(
|
||
f"{source.parent.name}/{source.name} : {changed}/{len(records)} page(s) rognée(s)",
|
||
flush=True,
|
||
)
|
||
return records
|
||
|
||
|
||
def process_files(
|
||
workspace: EvaluationWorkspace,
|
||
files: list[Path],
|
||
staging: Path,
|
||
workers: int,
|
||
) -> list[dict]:
|
||
heights: dict[str, float] = {}
|
||
for copy_name in sorted({source.parent.name for source in files}):
|
||
copy_pdf = workspace.copies_dir / f"{copy_name}.pdf"
|
||
if not copy_pdf.is_file():
|
||
raise CliError(f"PDF source introuvable : {copy_pdf}")
|
||
heights[copy_name] = _full_page_height(copy_pdf)
|
||
jobs = [
|
||
(
|
||
source,
|
||
staging / source.relative_to(workspace.copies_dir),
|
||
heights[source.parent.name],
|
||
)
|
||
for source in files
|
||
]
|
||
count = min(workers, len(jobs))
|
||
print(
|
||
f"Analyse de {len(files)} PDF de réponses avec {count} traitement(s) en parallèle.",
|
||
flush=True,
|
||
)
|
||
if count == 1:
|
||
previous_threads = cv2.getNumThreads()
|
||
cv2.setNumThreads(1)
|
||
try:
|
||
batches = [_process_file(job) for job in jobs]
|
||
finally:
|
||
cv2.setNumThreads(previous_threads)
|
||
else:
|
||
with multiprocessing.get_context("spawn").Pool(
|
||
count, _initialize_worker
|
||
) as pool:
|
||
batches = list(pool.imap_unordered(_process_file, jobs))
|
||
order = {source.as_posix(): index for index, source in enumerate(files)}
|
||
return sorted(
|
||
(record for batch in batches for record in batch),
|
||
key=lambda record: (order[record["file"]], record["page"]),
|
||
)
|
||
|
||
|
||
def _publish(
|
||
workspace: EvaluationWorkspace,
|
||
changed_files: list[Path],
|
||
staging: Path,
|
||
records: list[dict],
|
||
) -> Path:
|
||
workspace.runs_dir.mkdir(parents=True, exist_ok=True)
|
||
run_dir = Path(
|
||
tempfile.mkdtemp(prefix="crop-exercise-bottoms-", dir=workspace.runs_dir)
|
||
)
|
||
backup_root = run_dir / "Copies"
|
||
replaced: list[Path] = []
|
||
try:
|
||
for source in changed_files:
|
||
backup = backup_root / source.relative_to(workspace.copies_dir)
|
||
backup.parent.mkdir(parents=True, exist_ok=True)
|
||
shutil.copy2(source, backup)
|
||
(run_dir / "report.json").write_text(
|
||
json.dumps(records, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
|
||
)
|
||
for source in changed_files:
|
||
prepared = staging / source.relative_to(workspace.copies_dir)
|
||
prepared.replace(source)
|
||
replaced.append(source)
|
||
except BaseException:
|
||
for source in replaced:
|
||
backup = backup_root / source.relative_to(workspace.copies_dir)
|
||
if backup.is_file():
|
||
shutil.copy2(backup, source)
|
||
shutil.rmtree(run_dir, ignore_errors=True)
|
||
raise
|
||
return backup_root
|
||
|
||
|
||
def crop_statistics(records: list[dict]) -> tuple[int, float]:
|
||
"""Return cropped exercise count and mean removed percentage per exercise."""
|
||
totals: dict[str, list[float]] = {}
|
||
cropped_files: set[str] = set()
|
||
for record in records:
|
||
total_height, removed_height = totals.setdefault(record["file"], [0.0, 0.0])
|
||
totals[record["file"]] = [
|
||
total_height + float(record["height_lines"]),
|
||
removed_height + float(record["bottom_removed_lines"]),
|
||
]
|
||
if record["status"] == "cropped":
|
||
cropped_files.add(record["file"])
|
||
percentages = [
|
||
min(100.0, totals[file_name][1] / totals[file_name][0] * 100)
|
||
for file_name in cropped_files
|
||
if totals[file_name][0] > 0
|
||
]
|
||
mean_percentage = sum(percentages) / len(percentages) if percentages else 0.0
|
||
return len(cropped_files), mean_percentage
|
||
|
||
|
||
def run(workspace: EvaluationWorkspace, target: Path, *, workers: int = 5) -> ExitCode:
|
||
if workers < 1:
|
||
raise CliError(
|
||
"Le nombre de traitements parallèles doit être positif.",
|
||
ExitCode.INVALID_ARGUMENTS,
|
||
)
|
||
files = selected_files(workspace, target)
|
||
if not files:
|
||
raise CliError(
|
||
"Aucun PDF de réponse trouvé dans Copies/CopieXX/.",
|
||
ExitCode.INVALID_ARGUMENTS,
|
||
)
|
||
with tempfile.TemporaryDirectory(
|
||
prefix=".crop-exercise-bottoms-", dir=workspace.root
|
||
) as directory:
|
||
staging = Path(directory)
|
||
records = process_files(workspace, files, staging, workers)
|
||
changed_names = {
|
||
record["file"] for record in records if record["status"] == "cropped"
|
||
}
|
||
changed_files = [source for source in files if source.as_posix() in changed_names]
|
||
digests = {record["file"]: record["source_sha256"] for record in records}
|
||
for source in files:
|
||
if hashlib.sha256(source.read_bytes()).hexdigest() != digests[source.as_posix()]:
|
||
raise CliError(
|
||
f"{source} a changé pendant l’analyse. Aucun PDF remplacé."
|
||
)
|
||
cropped_exercises, mean_percentage = crop_statistics(records)
|
||
if not changed_files:
|
||
print("Terminé : aucun exercice ne remplit les critères de rognage.", flush=True)
|
||
print(
|
||
"Rognage moyen des exercices modifiés : 0.0 %.", flush=True
|
||
)
|
||
return ExitCode.SUCCESS
|
||
backup = _publish(workspace, changed_files, staging, records)
|
||
cropped_pages = sum(record["status"] == "cropped" for record in records)
|
||
print(f"Sauvegarde des PDF non rognés : {backup}", flush=True)
|
||
print(
|
||
f"Terminé : {cropped_exercises} exercice(s) rogné(s), soit "
|
||
f"{cropped_pages} page(s) dans {len(changed_files)} PDF remplacé(s).",
|
||
flush=True,
|
||
)
|
||
print(
|
||
f"Rognage moyen des exercices modifiés : {mean_percentage:.1f} %.",
|
||
flush=True,
|
||
)
|
||
return ExitCode.SUCCESS
|
||
|
||
|
||
def main(argv: Sequence[str] | None = None) -> int:
|
||
parser = target_parser(
|
||
"Rogner les grands espaces vides au bas des réponses déjà découpées"
|
||
)
|
||
parser.add_argument(
|
||
"--workers",
|
||
type=int,
|
||
default=5,
|
||
help="Nombre de PDF traités en parallèle (défaut : 5)",
|
||
)
|
||
|
||
def handle(arguments: argparse.Namespace) -> ExitCode:
|
||
workspace, target = workspace_from_target(arguments)
|
||
return run(workspace, target, workers=arguments.workers)
|
||
|
||
return execute(parser, argv, handle)
|
||
|
||
|
||
if __name__ == "__main__":
|
||
raise SystemExit(main())
|