Files
Copies/copienator/crop_exercise_bottoms.py
2026-09-09 15:24:39 +02:00

299 lines
11 KiB
Python

"""Propose large bottom-only crops for PDFs already split into exercises."""
from __future__ import annotations
import argparse
import csv
import hashlib
import html
import json
import multiprocessing
import signal
from pathlib import Path
from urllib.parse import quote
import cv2
import numpy as np
import pymupdf
from PIL import Image, ImageDraw
from copienator.crop_blank_margins import apply_bounds
from copienator.filesystem import staged_directory
from copienator.ink_detection import detect_bounds
LINES_PER_PAGE = 36
MINIMUM_HEIGHT_LINES = 10
IGNORED_BOTTOM_LINES = 0.75
MINIMUM_CROP_LINES = 4
def _displayed_media_height(page: pymupdf.Page) -> float:
"""Return the uncropped sheet height in the page's displayed orientation."""
return page.mediabox.width if page.rotation % 180 else page.mediabox.height
def _full_page_height(copy_pdf: Path) -> float:
with pymupdf.open(copy_pdf) as document:
if not len(document):
raise ValueError(f"PDF sans page : {copy_pdf}")
return max(_displayed_media_height(page) for page in document)
def _save_review(rgb: np.ndarray, bottom: int, destination: Path) -> None:
preview = Image.fromarray(rgb)
preview.thumbnail((500, 700))
overlay = Image.new("RGBA", preview.size)
draw = ImageDraw.Draw(overlay)
y = bottom / rgb.shape[0] * preview.height
draw.rectangle((0, y, preview.width, preview.height), fill=(255, 40, 40, 85))
draw.line((0, y, preview.width, y), fill=(230, 0, 0, 255), width=2)
destination.parent.mkdir(parents=True, exist_ok=True)
Image.alpha_composite(preview.convert("RGBA"), overlay).convert("RGB").save(
destination, quality=88
)
def process_exercise_pdf(
source: Path,
destination: Path,
review_dir: Path | None,
full_page_height: float,
*,
dpi: int = 200,
padding_mm: float = 6,
) -> list[dict]:
"""Crop qualifying pages and save the PDF only when at least one changes."""
digest = hashlib.sha256(source.read_bytes()).hexdigest()
line_points = full_page_height / LINES_PER_PAGE
records: list[dict] = []
changed = False
with pymupdf.open(source) as document:
page_count = len(document)
for index, page in enumerate(document):
visible_height = page.rect.height
record = {
"file": source.as_posix(),
"page": index + 1,
"page_count": page_count,
"source_sha256": digest,
"height_lines": round(visible_height / line_points, 2),
"bottom_removed_lines": 0.0,
"bottom_removed_mm": 0.0,
"status": "skipped-short",
}
if visible_height + 1e-6 < MINIMUM_HEIGHT_LINES * line_points:
records.append(record)
continue
pixmap = page.get_pixmap(
dpi=dpi, colorspace=pymupdf.csRGB, alpha=False
)
rgb = np.frombuffer(pixmap.samples, np.uint8).reshape(
pixmap.height, pixmap.width, 3
)
pixels_per_point = pixmap.height / visible_height
ignored_pixels = min(
pixmap.height - 1,
round(IGNORED_BOTTOM_LINES * line_points * pixels_per_point),
)
analysis_bottom = pixmap.height - ignored_pixels
detection = detect_bounds(
rgb[:analysis_bottom], dpi=dpi, padding_mm=padding_mm, min_crop_mm=0
)
proposed_bottom = detection["bottom_px"]
removed_points = (pixmap.height - proposed_bottom) / pixels_per_point
removed_lines = removed_points / line_points
record["detector_status"] = detection["status"]
record["proposed_bottom_removed_lines"] = round(removed_lines, 2)
if removed_lines + 1e-6 < MINIMUM_CROP_LINES:
record["status"] = "unchanged-small-crop"
records.append(record)
continue
apply_bounds(page, 0, proposed_bottom / pixmap.height)
record.update(
bottom_removed_lines=round(removed_lines, 2),
bottom_removed_mm=round(removed_points * 25.4 / 72, 2),
status="cropped",
output_height_points=round(page.rect.height, 3),
)
if review_dir is not None:
record["output"] = destination.relative_to(review_dir.parent).as_posix()
review_path = review_dir / source.parent.name / (
f"{source.stem}-p{index + 1:02}.jpg"
)
_save_review(rgb, proposed_bottom, review_path)
record["review"] = review_path.relative_to(review_dir.parent).as_posix()
changed = True
records.append(record)
if changed:
destination.parent.mkdir(parents=True, exist_ok=True)
document.save(destination, garbage=3, deflate=True)
if changed:
with pymupdf.open(destination) as check:
if len(check) != page_count:
raise RuntimeError(f"Nombre de pages modifié : {source}")
for page in check:
if page.rect.is_empty:
raise RuntimeError(f"Page vide produite : {destination}")
page.get_pixmap(matrix=pymupdf.Matrix(0.25, 0.25))
if hashlib.sha256(source.read_bytes()).hexdigest() != digest:
raise RuntimeError(f"PDF source modifié pendant le rognage : {source}")
return records
def _initialize_worker() -> None:
cv2.setNumThreads(1)
signal.signal(signal.SIGINT, signal.SIG_IGN)
def _process_job(job: tuple[Path, Path, Path, float, int, float]) -> list[dict]:
source, destination, review_dir, full_height, dpi, padding = job
records = process_exercise_pdf(
source, destination, review_dir, full_height, dpi=dpi, padding_mm=padding
)
changed = sum(record["status"] == "cropped" for record in records)
print(f"{source.parent.name}/{source.name} : {changed}/{len(records)} page(s) rognée(s)",
flush=True)
return records
def _write_index(output: Path, records: list[dict]) -> None:
changed = [record for record in records if record["status"] == "cropped"]
changed_files: dict[str, list[int]] = {}
for record in changed:
changed_files.setdefault(record["file"], []).append(record["page"])
(output / "cropped-files.txt").write_text(
"".join(
f"{path} - page(s) {', '.join(map(str, pages))}\n"
for path, pages in changed_files.items()
),
encoding="utf-8",
)
cards = []
for record in changed:
relative = Path(record["file"])
output_pdf = quote(record["output"])
preview = quote(record["review"])
name = html.escape(f"{relative.parent.name}/{relative.name} - page {record['page']}")
cards.append(
f'<article><a href="{output_pdf}#page={record["page"]}">{name}</a>'
f'<p>{record["bottom_removed_lines"]:.2f} lignes '
f'({record["bottom_removed_mm"]:.1f} mm) retirées</p>'
f'<img loading="lazy" src="{preview}"></article>'
)
(output / "index.html").write_text(
'<!doctype html><meta charset="utf-8"><title>Rognage bas des exercices</title>'
'<style>body{font:15px system-ui;background:#eee;margin:24px}'
'main{display:grid;grid-template-columns:repeat(auto-fill,minmax(330px,1fr));gap:20px}'
'article{background:white;padding:12px}img{width:100%}p{font-size:12px}</style>'
f'<h1>{len(changed)} pages rognées</h1>'
'<p>Le rouge montre la zone retirée. Seuls les PDF modifiés sont présents dans cropped/.</p>'
'<main>' + ''.join(cards) + '</main>',
encoding="utf-8",
)
def run(
input_path: Path,
output: Path,
*,
dpi: int = 200,
padding_mm: float = 6,
workers: int = 5,
) -> list[dict]:
copies = input_path / "Copies" if (input_path / "Copies").is_dir() else input_path
if not copies.is_dir():
raise ValueError(f"Dossier Copies introuvable : {input_path}")
sources = sorted(
copies.glob("Copie*/*.pdf"),
key=lambda path: (path.parent.name.casefold(), path.name.casefold()),
)
if not sources:
raise ValueError(f"Aucun PDF d'exercice trouvé dans {copies}")
if workers < 1:
raise ValueError("Le nombre de traitements parallèles doit être positif")
heights: dict[str, float] = {}
for copy_name in sorted({source.parent.name for source in sources}):
copy_pdf = copies / f"{copy_name}.pdf"
if not copy_pdf.is_file():
raise ValueError(f"PDF source introuvable : {copy_pdf}")
heights[copy_name] = _full_page_height(copy_pdf)
with staged_directory(output) as staging:
cropped_dir = staging / "cropped"
review_dir = staging / "review"
jobs = [
(
source,
cropped_dir / source.relative_to(copies),
review_dir,
heights[source.parent.name],
dpi,
padding_mm,
)
for source in sources
]
count = min(workers, len(jobs))
if count == 1:
previous_threads = cv2.getNumThreads()
cv2.setNumThreads(1)
try:
batches = [_process_job(job) for job in jobs]
finally:
cv2.setNumThreads(previous_threads)
else:
with multiprocessing.get_context("spawn").Pool(
count, _initialize_worker
) as pool:
batches = list(pool.imap_unordered(_process_job, jobs))
order = {source.as_posix(): i for i, source in enumerate(sources)}
records = sorted(
(record for batch in batches for record in batch),
key=lambda record: (order[record["file"]], record["page"]),
)
(staging / "report.json").write_text(
json.dumps(records, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
)
with (staging / "report.csv").open("w", encoding="utf-8", newline="") as stream:
fields = sorted({key for record in records for key in record})
writer = csv.DictWriter(stream, fieldnames=fields)
writer.writeheader()
writer.writerows(records)
_write_index(staging, records)
return records
def main() -> None:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("input", type=Path, help="Évaluation ou dossier Copies")
parser.add_argument("output", type=Path)
parser.add_argument("--dpi", type=int, default=200)
parser.add_argument("--padding-mm", type=float, default=6)
parser.add_argument("--workers", type=int, default=5)
arguments = parser.parse_args()
if arguments.dpi < 100 or arguments.padding_mm < 0:
parser.error("Utilisez dpi >= 100 et une marge positive ou nulle")
try:
records = run(
arguments.input,
arguments.output,
dpi=arguments.dpi,
padding_mm=arguments.padding_mm,
workers=arguments.workers,
)
except ValueError as error:
parser.error(str(error))
cropped = sum(record["status"] == "cropped" for record in records)
files = len({record["file"] for record in records if record["status"] == "cropped"})
print(f"Terminé : {cropped} page(s) dans {files} PDF rognée(s). Revue : "
f"{arguments.output / 'index.html'}")
if __name__ == "__main__":
main()