299 lines
11 KiB
Python
299 lines
11 KiB
Python
"""Propose large bottom-only crops for PDFs already split into exercises."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import csv
|
|
import hashlib
|
|
import html
|
|
import json
|
|
import multiprocessing
|
|
import signal
|
|
from pathlib import Path
|
|
from urllib.parse import quote
|
|
|
|
import cv2
|
|
import numpy as np
|
|
import pymupdf
|
|
from PIL import Image, ImageDraw
|
|
|
|
from copienator.crop_blank_margins import apply_bounds
|
|
from copienator.filesystem import staged_directory
|
|
from copienator.ink_detection import detect_bounds
|
|
|
|
|
|
LINES_PER_PAGE = 36
|
|
MINIMUM_HEIGHT_LINES = 10
|
|
IGNORED_BOTTOM_LINES = 0.75
|
|
MINIMUM_CROP_LINES = 4
|
|
|
|
|
|
def _displayed_media_height(page: pymupdf.Page) -> float:
|
|
"""Return the uncropped sheet height in the page's displayed orientation."""
|
|
return page.mediabox.width if page.rotation % 180 else page.mediabox.height
|
|
|
|
|
|
def _full_page_height(copy_pdf: Path) -> float:
|
|
with pymupdf.open(copy_pdf) as document:
|
|
if not len(document):
|
|
raise ValueError(f"PDF sans page : {copy_pdf}")
|
|
return max(_displayed_media_height(page) for page in document)
|
|
|
|
|
|
def _save_review(rgb: np.ndarray, bottom: int, destination: Path) -> None:
|
|
preview = Image.fromarray(rgb)
|
|
preview.thumbnail((500, 700))
|
|
overlay = Image.new("RGBA", preview.size)
|
|
draw = ImageDraw.Draw(overlay)
|
|
y = bottom / rgb.shape[0] * preview.height
|
|
draw.rectangle((0, y, preview.width, preview.height), fill=(255, 40, 40, 85))
|
|
draw.line((0, y, preview.width, y), fill=(230, 0, 0, 255), width=2)
|
|
destination.parent.mkdir(parents=True, exist_ok=True)
|
|
Image.alpha_composite(preview.convert("RGBA"), overlay).convert("RGB").save(
|
|
destination, quality=88
|
|
)
|
|
|
|
|
|
def process_exercise_pdf(
|
|
source: Path,
|
|
destination: Path,
|
|
review_dir: Path | None,
|
|
full_page_height: float,
|
|
*,
|
|
dpi: int = 200,
|
|
padding_mm: float = 6,
|
|
) -> list[dict]:
|
|
"""Crop qualifying pages and save the PDF only when at least one changes."""
|
|
digest = hashlib.sha256(source.read_bytes()).hexdigest()
|
|
line_points = full_page_height / LINES_PER_PAGE
|
|
records: list[dict] = []
|
|
changed = False
|
|
with pymupdf.open(source) as document:
|
|
page_count = len(document)
|
|
for index, page in enumerate(document):
|
|
visible_height = page.rect.height
|
|
record = {
|
|
"file": source.as_posix(),
|
|
"page": index + 1,
|
|
"page_count": page_count,
|
|
"source_sha256": digest,
|
|
"height_lines": round(visible_height / line_points, 2),
|
|
"bottom_removed_lines": 0.0,
|
|
"bottom_removed_mm": 0.0,
|
|
"status": "skipped-short",
|
|
}
|
|
if visible_height + 1e-6 < MINIMUM_HEIGHT_LINES * line_points:
|
|
records.append(record)
|
|
continue
|
|
|
|
pixmap = page.get_pixmap(
|
|
dpi=dpi, colorspace=pymupdf.csRGB, alpha=False
|
|
)
|
|
rgb = np.frombuffer(pixmap.samples, np.uint8).reshape(
|
|
pixmap.height, pixmap.width, 3
|
|
)
|
|
pixels_per_point = pixmap.height / visible_height
|
|
ignored_pixels = min(
|
|
pixmap.height - 1,
|
|
round(IGNORED_BOTTOM_LINES * line_points * pixels_per_point),
|
|
)
|
|
analysis_bottom = pixmap.height - ignored_pixels
|
|
detection = detect_bounds(
|
|
rgb[:analysis_bottom], dpi=dpi, padding_mm=padding_mm, min_crop_mm=0
|
|
)
|
|
proposed_bottom = detection["bottom_px"]
|
|
removed_points = (pixmap.height - proposed_bottom) / pixels_per_point
|
|
removed_lines = removed_points / line_points
|
|
record["detector_status"] = detection["status"]
|
|
record["proposed_bottom_removed_lines"] = round(removed_lines, 2)
|
|
if removed_lines + 1e-6 < MINIMUM_CROP_LINES:
|
|
record["status"] = "unchanged-small-crop"
|
|
records.append(record)
|
|
continue
|
|
|
|
apply_bounds(page, 0, proposed_bottom / pixmap.height)
|
|
record.update(
|
|
bottom_removed_lines=round(removed_lines, 2),
|
|
bottom_removed_mm=round(removed_points * 25.4 / 72, 2),
|
|
status="cropped",
|
|
output_height_points=round(page.rect.height, 3),
|
|
)
|
|
if review_dir is not None:
|
|
record["output"] = destination.relative_to(review_dir.parent).as_posix()
|
|
review_path = review_dir / source.parent.name / (
|
|
f"{source.stem}-p{index + 1:02}.jpg"
|
|
)
|
|
_save_review(rgb, proposed_bottom, review_path)
|
|
record["review"] = review_path.relative_to(review_dir.parent).as_posix()
|
|
changed = True
|
|
records.append(record)
|
|
|
|
if changed:
|
|
destination.parent.mkdir(parents=True, exist_ok=True)
|
|
document.save(destination, garbage=3, deflate=True)
|
|
|
|
if changed:
|
|
with pymupdf.open(destination) as check:
|
|
if len(check) != page_count:
|
|
raise RuntimeError(f"Nombre de pages modifié : {source}")
|
|
for page in check:
|
|
if page.rect.is_empty:
|
|
raise RuntimeError(f"Page vide produite : {destination}")
|
|
page.get_pixmap(matrix=pymupdf.Matrix(0.25, 0.25))
|
|
if hashlib.sha256(source.read_bytes()).hexdigest() != digest:
|
|
raise RuntimeError(f"PDF source modifié pendant le rognage : {source}")
|
|
return records
|
|
|
|
|
|
def _initialize_worker() -> None:
|
|
cv2.setNumThreads(1)
|
|
signal.signal(signal.SIGINT, signal.SIG_IGN)
|
|
|
|
|
|
def _process_job(job: tuple[Path, Path, Path, float, int, float]) -> list[dict]:
|
|
source, destination, review_dir, full_height, dpi, padding = job
|
|
records = process_exercise_pdf(
|
|
source, destination, review_dir, full_height, dpi=dpi, padding_mm=padding
|
|
)
|
|
changed = sum(record["status"] == "cropped" for record in records)
|
|
print(f"{source.parent.name}/{source.name} : {changed}/{len(records)} page(s) rognée(s)",
|
|
flush=True)
|
|
return records
|
|
|
|
|
|
def _write_index(output: Path, records: list[dict]) -> None:
|
|
changed = [record for record in records if record["status"] == "cropped"]
|
|
changed_files: dict[str, list[int]] = {}
|
|
for record in changed:
|
|
changed_files.setdefault(record["file"], []).append(record["page"])
|
|
(output / "cropped-files.txt").write_text(
|
|
"".join(
|
|
f"{path} - page(s) {', '.join(map(str, pages))}\n"
|
|
for path, pages in changed_files.items()
|
|
),
|
|
encoding="utf-8",
|
|
)
|
|
cards = []
|
|
for record in changed:
|
|
relative = Path(record["file"])
|
|
output_pdf = quote(record["output"])
|
|
preview = quote(record["review"])
|
|
name = html.escape(f"{relative.parent.name}/{relative.name} - page {record['page']}")
|
|
cards.append(
|
|
f'<article><a href="{output_pdf}#page={record["page"]}">{name}</a>'
|
|
f'<p>{record["bottom_removed_lines"]:.2f} lignes '
|
|
f'({record["bottom_removed_mm"]:.1f} mm) retirées</p>'
|
|
f'<img loading="lazy" src="{preview}"></article>'
|
|
)
|
|
(output / "index.html").write_text(
|
|
'<!doctype html><meta charset="utf-8"><title>Rognage bas des exercices</title>'
|
|
'<style>body{font:15px system-ui;background:#eee;margin:24px}'
|
|
'main{display:grid;grid-template-columns:repeat(auto-fill,minmax(330px,1fr));gap:20px}'
|
|
'article{background:white;padding:12px}img{width:100%}p{font-size:12px}</style>'
|
|
f'<h1>{len(changed)} pages rognées</h1>'
|
|
'<p>Le rouge montre la zone retirée. Seuls les PDF modifiés sont présents dans cropped/.</p>'
|
|
'<main>' + ''.join(cards) + '</main>',
|
|
encoding="utf-8",
|
|
)
|
|
|
|
|
|
def run(
|
|
input_path: Path,
|
|
output: Path,
|
|
*,
|
|
dpi: int = 200,
|
|
padding_mm: float = 6,
|
|
workers: int = 5,
|
|
) -> list[dict]:
|
|
copies = input_path / "Copies" if (input_path / "Copies").is_dir() else input_path
|
|
if not copies.is_dir():
|
|
raise ValueError(f"Dossier Copies introuvable : {input_path}")
|
|
sources = sorted(
|
|
copies.glob("Copie*/*.pdf"),
|
|
key=lambda path: (path.parent.name.casefold(), path.name.casefold()),
|
|
)
|
|
if not sources:
|
|
raise ValueError(f"Aucun PDF d'exercice trouvé dans {copies}")
|
|
if workers < 1:
|
|
raise ValueError("Le nombre de traitements parallèles doit être positif")
|
|
|
|
heights: dict[str, float] = {}
|
|
for copy_name in sorted({source.parent.name for source in sources}):
|
|
copy_pdf = copies / f"{copy_name}.pdf"
|
|
if not copy_pdf.is_file():
|
|
raise ValueError(f"PDF source introuvable : {copy_pdf}")
|
|
heights[copy_name] = _full_page_height(copy_pdf)
|
|
|
|
with staged_directory(output) as staging:
|
|
cropped_dir = staging / "cropped"
|
|
review_dir = staging / "review"
|
|
jobs = [
|
|
(
|
|
source,
|
|
cropped_dir / source.relative_to(copies),
|
|
review_dir,
|
|
heights[source.parent.name],
|
|
dpi,
|
|
padding_mm,
|
|
)
|
|
for source in sources
|
|
]
|
|
count = min(workers, len(jobs))
|
|
if count == 1:
|
|
previous_threads = cv2.getNumThreads()
|
|
cv2.setNumThreads(1)
|
|
try:
|
|
batches = [_process_job(job) for job in jobs]
|
|
finally:
|
|
cv2.setNumThreads(previous_threads)
|
|
else:
|
|
with multiprocessing.get_context("spawn").Pool(
|
|
count, _initialize_worker
|
|
) as pool:
|
|
batches = list(pool.imap_unordered(_process_job, jobs))
|
|
order = {source.as_posix(): i for i, source in enumerate(sources)}
|
|
records = sorted(
|
|
(record for batch in batches for record in batch),
|
|
key=lambda record: (order[record["file"]], record["page"]),
|
|
)
|
|
(staging / "report.json").write_text(
|
|
json.dumps(records, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
|
|
)
|
|
with (staging / "report.csv").open("w", encoding="utf-8", newline="") as stream:
|
|
fields = sorted({key for record in records for key in record})
|
|
writer = csv.DictWriter(stream, fieldnames=fields)
|
|
writer.writeheader()
|
|
writer.writerows(records)
|
|
_write_index(staging, records)
|
|
return records
|
|
|
|
|
|
def main() -> None:
|
|
parser = argparse.ArgumentParser(description=__doc__)
|
|
parser.add_argument("input", type=Path, help="Évaluation ou dossier Copies")
|
|
parser.add_argument("output", type=Path)
|
|
parser.add_argument("--dpi", type=int, default=200)
|
|
parser.add_argument("--padding-mm", type=float, default=6)
|
|
parser.add_argument("--workers", type=int, default=5)
|
|
arguments = parser.parse_args()
|
|
if arguments.dpi < 100 or arguments.padding_mm < 0:
|
|
parser.error("Utilisez dpi >= 100 et une marge positive ou nulle")
|
|
try:
|
|
records = run(
|
|
arguments.input,
|
|
arguments.output,
|
|
dpi=arguments.dpi,
|
|
padding_mm=arguments.padding_mm,
|
|
workers=arguments.workers,
|
|
)
|
|
except ValueError as error:
|
|
parser.error(str(error))
|
|
cropped = sum(record["status"] == "cropped" for record in records)
|
|
files = len({record["file"] for record in records if record["status"] == "cropped"})
|
|
print(f"Terminé : {cropped} page(s) dans {files} PDF rognée(s). Revue : "
|
|
f"{arguments.output / 'index.html'}")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|