diff --git a/.gitignore b/.gitignore index 6c65a75..f23ec68 100644 --- a/.gitignore +++ b/.gitignore @@ -7,4 +7,5 @@ dist/ *.egg-info/ config.py .copienator-gui.json -.copienator/ \ No newline at end of file +.copienator/ +tmp/ diff --git a/copienator/commands/crop_margins.py b/copienator/commands/crop_margins.py new file mode 100644 index 0000000..646a444 --- /dev/null +++ b/copienator/commands/crop_margins.py @@ -0,0 +1,131 @@ +"""Optional preprocessing: replace split copies with ink-guided crops.""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import multiprocessing +import signal +import shutil +import tempfile +import cv2 +from collections.abc import Sequence +from pathlib import Path + +from copienator.cli import CliError, ExitCode, execute, target_parser, workspace_from_target +from copienator.crop_blank_margins import process_pdf +from copienator.filesystem import staged_files +from copienator.workspace import EvaluationWorkspace + + +def selected_files(workspace: EvaluationWorkspace, target: Path) -> list[Path]: + workspace.require_directories("Copies") + copies = workspace.copies_dir.resolve() + target = target.resolve() + if target.is_file(): + if target.parent != copies or target.suffix.lower() != ".pdf": + raise CliError("La cible doit être un PDF du dossier Copies.", ExitCode.INVALID_ARGUMENTS) + files = [target] + elif target in (workspace.root, copies): + files = sorted(copies.glob("*.pdf"), key=lambda path: path.name.casefold()) + else: + raise CliError("Cible attendue : évaluation, dossier Copies ou PDF dans Copies.", + ExitCode.INVALID_ARGUMENTS) + for source in files: + if source.is_symlink(): + raise CliError(f"Lien symbolique non pris en charge : {source}") + if source.with_suffix(".json").exists(): + raise CliError( + f"{source.name} possède déjà des coordonnées de labels. " + "Le rognage doit précéder leur détection. Pour reprendre le prétraitement, " + "mettez de côté le JSON associé, puis régénérez la découpe des marges et les labels. " + "Aucun PDF n’a été remplacé.") + return files + + +def _initialize_worker() -> None: + # Five copies should use five cores, not five OpenCV thread pools. MuPDF + # must also stay isolated in separate processes rather than Python threads. + cv2.setNumThreads(1) + signal.signal(signal.SIGINT, signal.SIG_IGN) + + +def _process_copy(job: tuple[Path, Path]) -> list[dict]: + source, destination = job + + def progress(page, total, row): + removed = row["top_removed_mm"]+row["bottom_removed_mm"] + print(f"{source.name} — Page {page}/{total} : {removed:.1f} mm retirés", flush=True) + + return process_pdf(source, destination, None, 200, 6, 5, progress=progress) + + +def process_copies(files: list[Path], staging: Path, workers: int) -> list[dict]: + jobs = [(source, staging/source.name) for source in files] + count = min(workers, len(jobs)) + print(f"Rognage de {len(files)} copies avec {count} traitement(s) en parallèle.", flush=True) + if count == 1: + previous_threads = cv2.getNumThreads() + cv2.setNumThreads(1) + try: + batches = [_process_copy(job) for job in jobs] + finally: + cv2.setNumThreads(previous_threads) + else: + # spawn works on Windows and avoids inheriting GUI/native-library state. + # Pool's context terminates and joins workers on errors or cancellation + # before staged_files removes the unpublished PDFs. + with multiprocessing.get_context("spawn").Pool(count, _initialize_worker) as pool: + batches = list(pool.imap_unordered(_process_copy, jobs)) + order = {source.name: i for i, source in enumerate(files)} + return sorted((row for batch in batches for row in batch), + key=lambda row: (order[row["file"]], row["page"])) + + +def run(workspace: EvaluationWorkspace, target: Path, *, workers: int = 5) -> ExitCode: + if workers < 1: + raise CliError("Le nombre de traitements parallèles doit être positif.", + ExitCode.INVALID_ARGUMENTS) + files = selected_files(workspace, target) + if not files: + raise CliError("Aucun PDF trouvé dans Copies.", ExitCode.INVALID_ARGUMENTS) + # Prepare the whole batch before replacing any copy. A detection failure or + # interruption leaves the working PDFs intact; commit errors roll back. + with staged_files(workspace.copies_dir) as staging: + records = process_copies(files, staging, workers) + # Detect edits made while the batch was being analysed, before saving + # backups or publishing results derived from an obsolete source. + digests = {row["file"]: row["source_sha256"] for row in records} + for source in files: + if hashlib.sha256(source.read_bytes()).hexdigest() != digests[source.name]: + raise CliError(f"{source.name} a changé pendant l’analyse. Aucun PDF remplacé.") + workspace.runs_dir.mkdir(parents=True, exist_ok=True) + backup = Path(tempfile.mkdtemp(prefix="crop-margins-", dir=workspace.runs_dir)) + originals = backup/"Copies" + originals.mkdir() + for source in files: + shutil.copy2(source, originals/source.name) + (backup/"report.json").write_text(json.dumps(records, ensure_ascii=False, indent=2), + encoding="utf-8") + print(f"Sauvegarde des PDF non rognés : {originals}", flush=True) + cropped = sum(row["top_removed_mm"]+row["bottom_removed_mm"] > 0 for row in records) + print(f"Terminé : {cropped}/{len(records)} pages rognées ; " + f"{len(files)} PDF remplacés dans Copies.", flush=True) + return ExitCode.SUCCESS + + +def main(argv: Sequence[str] | None = None) -> int: + parser = target_parser("Rogner les zones vides des PDF dans Copies, avant les labels") + parser.add_argument("--workers", type=int, default=5, + help="Nombre de copies traitées en parallèle (défaut : 5)") + + def handle(args: argparse.Namespace) -> ExitCode: + workspace, target = workspace_from_target(args) + return run(workspace, target, workers=args.workers) + + return execute(parser, argv, handle) + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/copienator/crop_blank_margins.py b/copienator/crop_blank_margins.py new file mode 100644 index 0000000..6eb7945 --- /dev/null +++ b/copienator/crop_blank_margins.py @@ -0,0 +1,137 @@ +"""Propose conservative top/bottom crops for scanned, optionally ruled PDFs. + +Run with ``python -m copienator.crop_blank_margins INPUT_DIR OUTPUT_DIR``. +Only PDFs directly in INPUT_DIR are processed. Originals are never modified. +Analysis is deskewed; output keeps the original scan and changes its CropBox. +This is a heuristic review utility, not a guarantee that a scan contains no ink. +""" + +from __future__ import annotations + +import argparse +import csv +import hashlib +import html +import json +from pathlib import Path + +import cv2 +import numpy as np +import pymupdf +from PIL import Image, ImageDraw + +from copienator.ink_detection import detect_bounds + + +def apply_bounds(page: pymupdf.Page, top: float, bottom: float) -> None: + """Apply fractional bounds in displayed orientation, respecting old CropBox.""" + rect = page.rect + visible = pymupdf.Rect(0, top*rect.height, rect.width, bottom*rect.height) + box = visible * page.derotation_matrix + box += (page.cropbox_position.x, page.cropbox_position.y, + page.cropbox_position.x, page.cropbox_position.y) + page.set_cropbox(box) + + +def process_pdf(source: Path, destination: Path, review: Path | None, + dpi: float, padding_mm: float, min_crop_mm: float, + progress=None) -> list[dict]: + digest = hashlib.sha256(source.read_bytes()).hexdigest() + rows = [] + with pymupdf.open(source) as doc: + for index, page in enumerate(doc): + pix = page.get_pixmap(dpi=round(dpi), colorspace=pymupdf.csRGB, alpha=False) + rgb = np.frombuffer(pix.samples, np.uint8).reshape(pix.height, pix.width, 3) + result = detect_bounds(rgb, dpi, padding_mm, min_crop_mm) + top, bottom = result.pop('top_px'), result.pop('bottom_px') + row = dict(file=source.name, page=index+1, **result, + top_removed_mm=round(top/pix.height*page.rect.height*25.4/72, 2), + bottom_removed_mm=round((pix.height-bottom)/pix.height*page.rect.height*25.4/72, 2), + original_cropbox=list(page.cropbox), rotation=page.rotation, + source_sha256=digest) + if review is not None: + preview = Image.fromarray(rgb) + preview.thumbnail((500, 700)) + overlay = Image.new('RGBA', preview.size) + draw = ImageDraw.Draw(overlay) + y0, y1 = top/pix.height*preview.height, bottom/pix.height*preview.height + if top: + draw.rectangle((0, 0, preview.width, y0), fill=(255, 40, 40, 85)) + draw.line((0, y0, preview.width, y0), fill=(230, 0, 0, 255), width=2) + if bottom < pix.height: + draw.rectangle((0, y1, preview.width, preview.height), fill=(255, 40, 40, 85)) + draw.line((0, y1, preview.width, y1), fill=(230, 0, 0, 255), width=2) + preview = Image.alpha_composite(preview.convert('RGBA'), overlay).convert('RGB') + preview.save(review/f'{source.stem}-{index+1:03}.jpg', quality=85) + if top or bottom < pix.height: + apply_bounds(page, top/pix.height, bottom/pix.height) + row['output_cropbox'] = list(page.cropbox) + rows.append(row) + if progress is not None: + progress(index+1, len(doc), row) + doc.save(destination, garbage=3, deflate=True) + with pymupdf.open(destination) as check: + if len(check) != len(rows): + raise RuntimeError(f'Page count changed: {source}') + for page in check: + if page.rect.is_empty: + raise RuntimeError(f'Empty output page: {destination}') + if hashlib.sha256(source.read_bytes()).hexdigest() != digest: + raise RuntimeError(f'Source changed during processing: {source}') + return rows + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument('input', type=Path) + parser.add_argument('output', type=Path) + parser.add_argument('--dpi', type=int, default=200) + parser.add_argument('--padding-mm', type=float, default=6) + parser.add_argument('--min-crop-mm', type=float, default=5) + args = parser.parse_args() + if args.dpi < 100 or args.padding_mm < 0 or args.min_crop_mm < 0: + parser.error('Use dpi >= 100 and nonnegative margins.') + sources = sorted(args.input.glob('*.pdf')) if args.input.is_dir() else [args.input] + if not sources or any(not p.is_file() for p in sources): + parser.error('No input PDFs found.') + if any(p.resolve() == (args.output/p.name).resolve() for p in sources): + parser.error('Output must not overwrite input PDFs.') + args.output.mkdir(parents=True, exist_ok=True) + review = args.output/'review' + review.mkdir(exist_ok=True) + cv2.setNumThreads(2) + rows = [] + for i, source in enumerate(sources): + batch = process_pdf(source, args.output/source.name, review, + args.dpi, args.padding_mm, args.min_crop_mm) + rows.extend(batch) + print(f'[{i+1}/{len(sources)}] {source.name}: {len(batch)} pages, ' + f'{sum(r["top_removed_mm"] > 0 or r["bottom_removed_mm"] > 0 for r in batch)} cropped', + flush=True) + (args.output/'report.json').write_text(json.dumps(rows, indent=2)+'\n') + with (args.output/'report.csv').open('w') as stream: + writer = csv.DictWriter(stream, fieldnames=list(rows[0])) + writer.writeheader() + writer.writerows(rows) + cards = [] + for row in rows: + stem = Path(row['file']).stem + cards.append(f'
' + f'{html.escape(stem)} / {row["page"]}' + f'

Top: {row["top_removed_mm"]} mm · Bottom: {row["bottom_removed_mm"]} mm' + f' · {row["status"]}

' + f'
') + (args.output/'index.html').write_text( + 'Crop review' + '' + '

Crop review

Red shading shows the removed areas on the original scan. ' + 'Click a page title to open the processed PDF. Review statuses flag uncertain detections.

' + '
'+''.join(cards)+'
') + changed = sum(r['top_removed_mm'] > 0 or r['bottom_removed_mm'] > 0 for r in rows) + print(f'Done: {len(sources)} PDFs, {len(rows)} pages, {changed} cropped. Review: {args.output / "index.html"}') + + +if __name__ == '__main__': + main() diff --git a/copienator/dispatcher.py b/copienator/dispatcher.py index 5c69357..a7decc5 100644 --- a/copienator/dispatcher.py +++ b/copienator/dispatcher.py @@ -17,6 +17,7 @@ COMMANDS: dict[str, Command] = { "statement-personal": Command("enonce_info", "Generate personal statement metadata"), "copies": Command("copies_tools", "Rotate or rename scanned copies"), "page-split": Command("page_splitter", "Split and reorder scanned PDF pages"), + "crop-margins": Command("crop_margins", "Trim blank top and bottom margins in Copies"), "crop-labels": Command("cutleft", "Crop the label margin from copies"), "labels": Command("gemini_for_labels", "Detect question labels with Gemini"), "review-labels": Command("plotting", "Review detected labels interactively"), diff --git a/copienator/ink_detection.py b/copienator/ink_detection.py new file mode 100644 index 0000000..357fb94 --- /dev/null +++ b/copienator/ink_detection.py @@ -0,0 +1,260 @@ +"""Detect top and bottom content bounds in scanned student work. + +Strong coloured strokes are never erased merely because they coincide with +paper ruling. The detector supports coloured handwriting and dark ruled scans; +it remains conservative for pencil-only scans and heavily saturated grids. +""" + +from functools import lru_cache + +import cv2 +import numpy as np + +from copienator.paper_background import _skew, _ruling, _foreground, _content_mask + + +@lru_cache(maxsize=8) +def _paper_kernel(sigma: float) -> np.ndarray: + """Match OpenCV's uint8 Gaussian coefficients, including error diffusion. + + See getGaussianKernelFixedPoint_ED in OpenCV's smooth.dispatch.cpp. + The 8-bit coefficients allow exact sums in the faster float filter path. + """ + size = round(sigma*6+1) | 1 + kernel = cv2.getGaussianKernel(size, sigma).ravel() + fixed = np.zeros(size, np.float32) + error = 0.0 + for i in range(size//2): + value = kernel[i]*256+error + weight = round(value) + error = value-weight + fixed[i] = fixed[-1-i] = weight + fixed[size//2] = 256-fixed.sum() + fixed /= 256 + return fixed + + +def _paper_blur(gray: np.ndarray, sigma: float) -> np.ndarray: + kernel = _paper_kernel(sigma) + blurred = cv2.sepFilter2D(gray, cv2.CV_32F, kernel, kernel) + # GaussianBlur rounds positive half-integers upward, rather than to even. + np.add(blurred, .5, out=blurred) + np.floor(blurred, out=blurred) + return blurred.astype(np.uint8) + + +def _large_blank_ink(gray: np.ndarray, clean: np.ndarray, dpi: float): + """Refine confirmed ruling, only accepting substantial extra blank margins. + + Short directional openings tolerate locally bent/broken paper lines. A + physical component-size threshold rejects their remaining tiny fragments. + This is deliberately limited to already-confirmed ruled paper. + """ + px = dpi / 25.4 + height, width = gray.shape + dark = 255-gray + length = max(9, round(2.5*px)) + tolerance = max(3, round(.6*px) | 1) + lines = [] + for horizontal in (True, False): + broadened = cv2.dilate(dark, np.ones( + (tolerance, 1) if horizontal else (1, tolerance), np.uint8)) + lines.append(cv2.morphologyEx(broadened, cv2.MORPH_OPEN, np.ones( + (1, length) if horizontal else (length, 1), np.uint8))) + residual = cv2.subtract(dark, np.maximum(*lines)) + n, labels, stats, centers = cv2.connectedComponentsWithStats( + (residual > 35).astype(np.uint8), 8) + keep = np.zeros(n, bool) + ww, hh, area = stats[1:, 2], stats[1:, 3], stats[1:, 4] + keep[1:] = ((area >= .8*px*px) & (np.minimum(ww, hh) >= .6*px) + & (area / (ww*hh) > .3) + & (np.maximum(ww, hh) < 4*np.minimum(ww, hh))) + # Use side columns as a prior, then require repeated size and alignment. An + # isolated note in the same column remains eligible to protect the margin. + xx = stats[1:, 0] + candidates = np.flatnonzero( + ((xx < 15*px) | (xx+ww > width-15*px)) + & (ww > px) & (ww < 9*px) & (hh > px) & (hh < 12*px)) + 1 + if len(candidates) > 128: + # Bound the matching cost and retain ambiguous, very noisy margins. + return clean, 0, False + holes = set() + for i in candidates: + matches = [] + for j in candidates: + if (abs(centers[i, 0]-centers[j, 0]) < 2*px + and .6 < stats[j, 2]/stats[i, 2] < 1.6 + and .6 < stats[j, 3]/stats[i, 3] < 1.6): + matches.append(j) + if len(matches) >= 3 and np.ptp(centers[matches, 1]) > height*.35: + holes.update(matches) + keep[list(holes)] = False + refined = keep[labels] + # Directional opening also removes long fraction bars. Protect very dark, + # thick straight strokes independently, even if ruling crosses their ends. + long_strokes = cv2.morphologyEx((gray < 50).astype(np.uint8), cv2.MORPH_OPEN, + np.ones((1, max(9, round(width*.1))), np.uint8)) + count, lab, st, _ = cv2.connectedComponentsWithStats(long_strokes, 8) + bars = np.zeros(count, bool) + bw, bh, ba = st[1:, 2], st[1:, 3], st[1:, 4] + bars[1:] = ((bw > width*.1) & (bh >= .3*px) & (bh < height*.015) + & (bw > 8*bh) & (ba/(bw*bh) > .5)) + strong_ruling, strong_lines = _ruling((gray < 50).astype(np.uint8)*255, True) + if strong_lines: + # A family of equally dark parallel lines is paper, not fraction bars. + overlap = np.bincount(lab[strong_ruling > 0], minlength=count) + bars &= overlap < st[:, 4]*.5 + refined |= bars[lab] + ys = np.flatnonzero(np.any(refined, axis=1)) + original = np.flatnonzero(np.any(clean, axis=1)) + if not len(ys) or not len(original): + return clean, 0, False + # Leave ordinary small crops to the more permissive detector. Keep a + # recovery neighbourhood around the refined bounds for broken/faint strokes. + top, bottom = max(0, int(ys.min()-2*px)), min(height, int(ys.max()+1+2*px)) + result = clean.copy() + changed = False + if top-original.min() >= 30*px: + result[:top] = 0 + changed = True + if original.max()+1-bottom >= 30*px: + result[bottom:] = 0 + changed = True + return result, len(holes), changed + + +def _neutral_paper_foreground(gray: np.ndarray, chroma: np.ndarray, dpi: float): + """Clean confirmed dark ruling before it can seed whole pages. + + Returns None for ordinary ink components or unconfirmed paper geometry. + The mask is transformed back to the original displayed pixel coordinates. + """ + height, width = gray.shape + # Only neutral darkness is relevant here: long blue equations are not + # evidence of dark paper. Broken ruling can form several medium-sized + # components instead of one page-spanning component. The broader threshold + # also admits faded gray grids; periodic ruling must still be confirmed. + _, _, stats, _ = cv2.connectedComponentsWithStats( + ((gray < 160) & (chroma < 30)).astype(np.uint8), 8) + spans = np.maximum(stats[1:,2]/width, stats[1:,3]/height) + if not (np.any(spans > .35) or np.count_nonzero(spans > .1) >= 3): + return None, dict(paper_cleanup=False) + angle = _skew(gray, angle_step=.5) + matrix = cv2.getRotationMatrix2D((width/2, height/2), angle, 1) + corners = np.array([[0,0],[width,0],[0,height],[width,height]], dtype=float) + corners = cv2.transform(corners[None], matrix)[0] + origin = np.floor(corners.min(axis=0)) + size = np.ceil(corners.max(axis=0)-origin).astype(int) + matrix[:, 2] -= origin + deskewed = cv2.warpAffine(gray, matrix, tuple(size), borderValue=255) + background = _paper_blur(deskewed, dpi/8) + normalized = cv2.divide(deskewed, np.maximum(background,1), scale=255) + block = max(15, int(dpi/5) | 1) + binary = cv2.adaptiveThreshold(normalized,255,cv2.ADAPTIVE_THRESH_GAUSSIAN_C, + cv2.THRESH_BINARY_INV,block,9) + horizontal, nh = _ruling(binary,True) + vertical, nv = _ruling(binary,False) + if not (nh or nv): + return None, dict(paper_cleanup=False) + # The permissive mask retains faint, isolated marks. + # Bands identify where ruling is expected, but only actual dark pixels + # inside them may be suppressed. Erasing the complete band loses faint + # writing alongside a dark grid line. + paper_pixels = (horizontal|vertical) & ((normalized < 160).astype(np.uint8)*255) + clean, holes = _content_mask(_foreground(normalized,dpi,9,paper_pixels),dpi,nh,nv) + clean, repeated_holes, refined = _large_blank_ink(deskewed, clean, dpi) + restored = cv2.warpAffine(clean, cv2.invertAffineTransform(matrix), + (width,height), flags=cv2.INTER_NEAREST, borderValue=0) + return restored > 0, dict(paper_cleanup=True, angle_deg=round(angle,3), + horizontal_lines=nh,vertical_lines=nv, + edge_artifacts=holes+repeated_holes, + large_blank_refinement=refined) + + +def _seed_mask(mask: np.ndarray, px: float) -> np.ndarray: + n, labels, stats, _ = cv2.connectedComponentsWithStats(mask.astype(np.uint8), 8) + keep = np.zeros(n, bool) + keep[1:] = ((stats[1:, 4] >= max(4, .12*px*px)) + & (stats[1:, 2] >= max(2, round(.35*px))) + & (stats[1:, 3] >= max(2, round(.35*px)))) + return keep[labels] + + +def detect_bounds(rgb: np.ndarray, dpi: float = 200, padding_mm: float = 6, + min_crop_mm: float = 5) -> dict: + """Locate strong ink, recover adjacent faint strokes, and retain padding. + + Neutral punched-hole shadows usually have neither sufficient chroma nor + sufficient darkness to seed a region. Nothing is discarded merely because + it is in a side margin. Two seed thresholds expose unstable boundaries. + """ + px = dpi/25.4 + h, w = rgb.shape[:2] + red, green, blue = cv2.split(rgb) + lowest = cv2.min(cv2.min(red, green), blue) + chroma = cv2.subtract(cv2.max(cv2.max(red, green), blue), lowest) + gray = cv2.cvtColor(rgb, cv2.COLOR_RGB2GRAY) + darkness = 255-lowest + length = max(15, round(5.2*px)) | 1 + horizontal = cv2.morphologyEx(darkness, cv2.MORPH_OPEN, + np.ones((1, length), np.uint8)) + vertical = cv2.morphologyEx(darkness, cv2.MORPH_OPEN, + np.ones((length, 1), np.uint8)) + residual = cv2.subtract(darkness, np.maximum(horizontal, vertical)) + # Strong strokes bypass the line-background estimate entirely: even a long + # isolated black or coloured fraction bar must survive. Local contrast is + # used only to recover weaker surrounding strokes. + neutral_ink = gray < 95 + cleaned, paper_info = _neutral_paper_foreground(gray, chroma, dpi) + if cleaned is not None: + neutral_ink = cleaned + weak = ((chroma > 60) | ((gray < 175) & (residual > 25))).astype(np.uint8) + radius = max(1, round(2*px)) + kernel = cv2.getStructuringElement(cv2.MORPH_ELLIPSE, (2*radius+1, 2*radius+1)) + extents = [] + counts = [] + same_thresholds = not np.any((chroma > 100) & (chroma <= 115) & ~neutral_ink) + for threshold in (100, 115): + if threshold == 115 and same_thresholds: + counts.append(counts[0]) + if extents: + extents.append(extents[0]) + continue + seeds = _seed_mask((chroma > threshold) | neutral_ink, px) + counts.append(int(np.count_nonzero(seeds))) + if not np.any(seeds): + continue + # Limit weak recovery to a physical neighbourhood: faint grid lines + # connected to a letter cannot grow into a full-page foreground mask. + nearby = cv2.dilate(seeds.astype(np.uint8), kernel) + candidate = (weak & nearby) | seeds.astype(np.uint8) + n, labels = cv2.connectedComponents(candidate, 8) + seeded_labels = np.zeros(n, bool) + seeded_labels[np.unique(labels[seeds])] = True + seeded_labels[0] = False + ys = np.flatnonzero(np.any(seeded_labels[labels], axis=1)) + extents.append((int(ys.min()), int(ys.max())+1)) + result = dict(top_px=0, bottom_px=h, angle_deg=None, + horizontal_lines=0, vertical_lines=0, edge_artifacts=0, + detector="ink", seed_pixels=counts, status="review-no-ink-seeds") + result.update(paper_info) + if not extents: + return result + pad = padding_mm*px + top = max(0, int(np.floor(min(e[0] for e in extents)-pad))) + bottom = min(h, int(np.ceil(max(e[1] for e in extents)+pad))) + uncertain = [] + if len(extents) < 2: + uncertain.append('seed-threshold') + else: + for edge, name in ((0, 'top'), (1, 'bottom')): + if abs(extents[0][edge]-extents[1][edge]) > max(pad, 3*px): + uncertain.append(name) + if top < min_crop_mm*px: + top = 0 + if h-bottom < min_crop_mm*px: + bottom = h + result.update(top_px=top, bottom_px=bottom, + status=('review-'+'-'.join(uncertain) if uncertain else + 'cropped' if top or bottom < h else 'unchanged')) + return result diff --git a/copienator/paper_background.py b/copienator/paper_background.py new file mode 100644 index 0000000..0d43241 --- /dev/null +++ b/copienator/paper_background.py @@ -0,0 +1,146 @@ +"""Shared geometry and foreground helpers for scanned paper backgrounds.""" + +import cv2 +import numpy as np + + +def _runs(values: np.ndarray) -> list[tuple[int, int]]: + edges = np.diff(np.r_[False, values, False].astype(np.int8)) + return list(zip(np.flatnonzero(edges == 1), np.flatnonzero(edges == -1))) + + +def _skew(gray: np.ndarray, angle_step: float = .1) -> float: + """Use the dominant near-horizontal/vertical Hough angle, at reduced size.""" + scale = min(1.0, 1200 / max(gray.shape)) + small = cv2.resize(gray, None, fx=scale, fy=scale) + edges = cv2.Canny(small, 40, 120) + lines = cv2.HoughLinesP(edges, 1, np.deg2rad(angle_step), 60, + minLineLength=min(small.shape) * .16, maxLineGap=12) + if lines is None: + return 0.0 + angles, weights = [], [] + for x0, y0, x1, y1 in lines.reshape(-1, 4): + a = (np.degrees(np.arctan2(y1-y0, x1-x0)) + 45) % 90 - 45 + if abs(a) <= 5: + angles.append(a) + weights.append(np.hypot(x1-x0, y1-y0)) + return _dominant_angle(angles, weights) + + +def _dominant_angle(angles, weights) -> float: + if len(angles) < 4: + return 0.0 + angles, weights = np.array(angles), np.array(weights) + bins = np.arange(-5.125, 5.126, .25) + hist, _ = np.histogram(angles, bins, weights=weights) + peak = (bins[hist.argmax()] + bins[hist.argmax()+1]) / 2 + near = abs(angles-peak) < .4 + if weights[near].sum() < .35 * weights.sum(): + return 0.0 + return float(np.average(angles[near], weights=weights[near])) + + +def _ruling(binary: np.ndarray, horizontal: bool) -> tuple[np.ndarray, int]: + """Accept a family of long lines only when positions are largely periodic.""" + h, w = binary.shape + length = max(25, int((w if horizontal else h) * .12)) + kernel = cv2.getStructuringElement(cv2.MORPH_RECT, + (length, 1) if horizontal else (1, length)) + connected = cv2.morphologyEx(binary, cv2.MORPH_CLOSE, + np.ones((1, 3) if horizontal else (3, 1), np.uint8)) + lines = cv2.morphologyEx(connected, cv2.MORPH_OPEN, kernel) + counts = np.count_nonzero(lines, axis=1 if horizontal else 0) + bands = _runs(counts > (w if horizontal else h) * .18) + if len(bands) < 5: + return np.zeros_like(binary), 0 + centers = np.array([(a+b)/2 for a, b in bands]) + gaps = np.diff(centers) + # Missing lines and major/minor rulings may have integer-multiple spacing. + candidates = gaps[gaps >= 4] + regular = any(np.mean(abs(gaps / d - np.round(gaps / d)) < .16) >= .75 + for d in candidates) + if not regular: + return np.zeros_like(binary), 0 + accepted = np.zeros_like(binary) + for a, b in bands: + if horizontal: + accepted[max(0, a-2):b+2] = 255 + else: + accepted[:, max(0, a-2):b+2] = 255 + # Real scans have local warp as well as global skew. Recover shorter line + # segments close to the established ruling family, without extending the + # entire family into large empty gaps. + short = max(25, int((w if horizontal else h)*.035)) + joined = cv2.morphologyEx(binary, cv2.MORPH_CLOSE, + np.ones((1, 7) if horizontal else (7, 1), np.uint8)) + tolerant = cv2.dilate(joined, np.ones((3, 1) if horizontal else (1, 3), np.uint8)) + fragments = cv2.morphologyEx(tolerant, cv2.MORPH_OPEN, + np.ones((1, short) if horizontal else (short, 1), np.uint8)) + nearby = cv2.dilate(accepted, np.ones((15, 1) if horizontal else (1, 15), np.uint8)) + accepted |= fragments & nearby if horizontal else fragments + return accepted, len(bands) + + +def _foreground(gray: np.ndarray, dpi: float, threshold: int, + ruling: np.ndarray) -> np.ndarray: + block = max(15, int(dpi / 5) | 1) + binary = cv2.adaptiveThreshold(gray, 255, cv2.ADAPTIVE_THRESH_GAUSSIAN_C, + cv2.THRESH_BINARY_INV, block, threshold) + # Only suppress near the fitted paper lines. Preserve unusually dark ink + # crossing pale ruling by comparing against the typical line intensity. + mask = cv2.dilate(ruling, np.ones((3, 3), np.uint8)) > 0 + if np.any(ruling): + samples = ((ruling > 0) & (binary > 0)).astype(np.float32) + window = max(31, int(dpi*.4) | 1) + weight = cv2.boxFilter(samples, -1, (window, window)) + total = cv2.boxFilter(gray.astype(np.float32)*samples, -1, (window, window)) + typical = total / np.maximum(weight, 1e-6) + excess_ink = (gray.astype(float) < typical - 40).astype(np.uint8) + # A dark, one-pixel remnant of a paper line is still paper. Only retain + # locally thicker excess strokes inside the suppression mask. + excess_ink = cv2.erode(excess_ink, np.ones((2, 2), np.uint8)) > 0 + mask &= ~excess_ink + binary[mask] = 0 + return binary + + +def _content_mask(binary: np.ndarray, dpi: float, nh: int = 0, + nv: int = 0) -> tuple[np.ndarray, int]: + """Filter only tiny speckles and repeated, matching edge-hole components.""" + px = dpi / 25.4 + # Small closing reconnects strokes interrupted by paper-line suppression. + grouped = cv2.morphologyEx(binary, cv2.MORPH_CLOSE, np.ones((3, 3), np.uint8)) + n, labels, stats, _ = cv2.connectedComponentsWithStats(grouped, 8) + keep = np.zeros(n, dtype=bool) + x, y, w, h, area = stats[1:].T + keep[1:] = (area >= max(4, .035*px*px)) & (np.maximum(w, h) >= .35*px) + # Thin straight residuals on confirmed ruled paper are not credible ink. + thin = max(2, round(.3*px)) + if nv: + keep[1:] &= ~((w <= thin) & (h >= 3*w)) + if nh: + keep[1:] &= ~((h <= thin) & (w >= 3*h)) + # Hole shadows form recurring shapes close to a physical side edge. Never + # discard an entire margin strip: other writing there must remain visible. + height, width = binary.shape + candidates = np.flatnonzero(keep[1:] + & ((x+w < 12*px) | (x > width-12*px)) + & (w > .8*px) & (w < 9*px) & (h > px) & (h < 12*px)) + 1 + holes = set() + for i in candidates: + x, y, w, h, area = stats[i] + similar = [] + a = cv2.resize((labels[y:y+h, x:x+w] == i).astype(np.uint8), (24, 32)) > 0 + for j in candidates: + xx, yy, ww, hh, aa = stats[j] + if abs(x-xx) > 2*px or not (.7 < ww/w < 1.4 and .7 < hh/h < 1.4): + continue + b = cv2.resize((labels[yy:yy+hh, xx:xx+ww] == j).astype(np.uint8), (24, 32)) > 0 + if np.count_nonzero(a & b) / max(1, np.count_nonzero(a | b)) > .60: + similar.append(j) + if len(similar) >= 4 and np.ptp(stats[similar, 1]) > height*.45: + holes.update(similar) + if holes: + keep[list(holes)] = False + # Bound original residual ink, not the expanded/grouped mask. + return (keep[labels] & (binary > 0)).astype(np.uint8), len(holes) diff --git a/copienator_gui/app.py b/copienator_gui/app.py index 22bf1fc..2fd0dce 100644 --- a/copienator_gui/app.py +++ b/copienator_gui/app.py @@ -589,7 +589,7 @@ class CopienatorApp(tk.Tk): return self._rendering = True redo = step.section == REFAIRE_SECTION - expanded_form = redo or step.id in {"page_splitter", "cutleft", "labels"} + expanded_form = redo or step.id in {"page_splitter", "crop_blank_margins", "cutleft", "labels"} self.console.configure(height=8 if expanded_form else 14) self.rowconfigure(1, weight=4 if expanded_form else 3) self.rowconfigure(2, weight=1 if expanded_form else 2) @@ -731,7 +731,7 @@ class CopienatorApp(tk.Tk): return row + 1 if step.id in {"review_persp", "correction"}: row = self._correction_folder_buttons(row) - if step.id in {"page_splitter", "cutleft", "labels"}: + if step.id in {"page_splitter", "crop_blank_margins", "cutleft", "labels"}: evaluation = self.evaluation paths = copy_pdf_paths(evaluation) if evaluation else [] self.copy_paths = {path.name: path for path in paths} @@ -762,7 +762,7 @@ class CopienatorApp(tk.Tk): command=self._open_selected_copy, state="normal" if names else "disabled", ).grid(row=1, column=1, padx=(6, 0), pady=(7, 0)) - if step.id in {"page_splitter", "cutleft", "labels"}: + if step.id in {"page_splitter", "crop_blank_margins", "cutleft", "labels"}: actions = ttk.Frame(copies) actions.grid(row=2, column=0, columnspan=2, sticky="w", pady=(7, 0)) label = "Refaire la copie sélectionnée" if step.id == "page_splitter" else "Cibler la copie sélectionnée" diff --git a/copienator_gui/workflow.py b/copienator_gui/workflow.py index 4cd5285..116d28c 100644 --- a/copienator_gui/workflow.py +++ b/copienator_gui/workflow.py @@ -176,6 +176,19 @@ def build_workflow(show_personal_steps: bool) -> list[StepDefinition]: arguments=(arg_target(), ArgumentSpec("marked", "Copies signalées uniquement", "bool", "--marked")), artifacts=("Copies", "Copies Originales"), ), + StepDefinition( + "crop_blank_margins", + "Prétraitement des copies", + "Rogner les zones vides", + "Facultatif : détecte les zones vides en haut et en bas malgré les lignes et les perforations, " + "puis remplace les PDF dans Copies. Les versions non rognées sont sauvegardées. " + "À effectuer avant la détection des labels. Traite plusieurs copies en parallèle.", + (python("default", "Rognage des zones vides", "crop-margins"),), + arguments=(arg_target("Dossier de l’évaluation ou PDF dans Copies"), + ArgumentSpec("workers", "Copies traitées en parallèle", "int", "--workers", default=5)), + optional=True, + requires=("Copies",), + ), StepDefinition( "cutleft", "Prétraitement des copies", @@ -235,16 +248,6 @@ def build_workflow(show_personal_steps: bool) -> list[StepDefinition]: artifacts=("Par label",), auto_start_first_visit=True, ), - StepDefinition( - "verify_groups", - "Labels et regroupement", - "Vérifier les groupes produits", - "Vérifie que chaque réponse PDF apparaît dans les métadonnées des groupes.", - (python("default", "Vérification des groupes", "verify-groups"),), - arguments=(arg_target("Dossier de l’évaluation"),), - optional=True, - requires=("Copies", "Par label"), - ), StepDefinition( "correction", "Correction", diff --git a/pyproject.toml b/pyproject.toml index b7d26f2..ff0fff8 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -9,6 +9,7 @@ description = "Workflow assistant for preparing and correcting scanned exams" requires-python = ">=3.11" dependencies = [ "numpy", + "opencv-python-headless>=4.8", "pandas", "matplotlib", "Pillow", diff --git a/tests/test_crop_blank_margins.py b/tests/test_crop_blank_margins.py new file mode 100644 index 0000000..ba0a9d4 --- /dev/null +++ b/tests/test_crop_blank_margins.py @@ -0,0 +1,30 @@ +import unittest + +import numpy as np +import pymupdf + +from copienator.crop_blank_margins import apply_bounds + + +class CropBlankMarginsTests(unittest.TestCase): + def test_cropbox_coordinates_with_rotation_and_existing_crop(self): + for rotation in (0, 90, 180, 270): + with self.subTest(rotation=rotation), pymupdf.open() as doc: + page = doc.new_page(width=600, height=800) + page.set_cropbox(pymupdf.Rect(30, 40, 570, 760)) + page.set_rotation(rotation) + before = page.rect + page.insert_text((80, 220), 'Visible content', fontsize=20) + pix = page.get_pixmap() + original = np.frombuffer(pix.samples, np.uint8).reshape(pix.height, pix.width, 3) + apply_bounds(page, .2, .8) + self.assertAlmostEqual(page.rect.width, before.width) + self.assertAlmostEqual(page.rect.height, before.height*.6, places=3) + self.assertEqual(page.rotation, rotation) + pix = page.get_pixmap() + cropped = np.frombuffer(pix.samples, np.uint8).reshape(pix.height, pix.width, 3) + np.testing.assert_array_equal(cropped, original[int(before.height*.2):int(before.height*.8)]) + + +if __name__ == '__main__': + unittest.main() diff --git a/tests/test_crop_margins_command.py b/tests/test_crop_margins_command.py new file mode 100644 index 0000000..1a8ef6f --- /dev/null +++ b/tests/test_crop_margins_command.py @@ -0,0 +1,189 @@ +import contextlib +import io +import os +import signal +import subprocess +import sys +import tempfile +import unittest +from pathlib import Path +from unittest.mock import patch + +import pymupdf + +from copienator.cli import CliError +from copienator.commands import crop_margins +from copienator.commands.clean import build_cleanup_plan, apply_cleanup +from copienator.dispatcher import main +from copienator.workspace import EvaluationWorkspace +from copienator_gui.workflow import build_workflow + + +class CropMarginsCommandTests(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory() + self.addCleanup(self.temp.cleanup) + self.workspace = EvaluationWorkspace(Path(self.temp.name)/"Évaluation") + self.workspace.copies_dir.mkdir(parents=True) + self.source = self.workspace.copies_dir/"Copie01.pdf" + with pymupdf.open() as doc: + page = doc.new_page(width=300, height=420) + page.insert_text((50, 180), "answer = 42", fontsize=15) + doc.new_page(width=300, height=420) + doc.save(self.source) + self.original = self.source.read_bytes() + + def test_dispatcher_replaces_copies_and_keeps_recoverable_original(self): + with contextlib.redirect_stdout(io.StringIO()) as log: + self.assertEqual(main(["crop-margins", str(self.workspace.root)]), 0) + self.assertIn("Page 2/2", log.getvalue()) + with pymupdf.open(self.source) as result: + self.assertEqual(len(result), 2) + self.assertLess(result[0].rect.height, 150) + self.assertIn("answer = 42", result[0].get_text()) + self.assertEqual(result[1].rect.height, 420) + for page in result: + self.assertEqual(page.rect.width, 300) + page.get_pixmap() + backups = list(self.workspace.runs_dir.glob("crop-margins-*/Copies/Copie01.pdf")) + self.assertEqual(len(backups), 1) + self.assertEqual(backups[0].read_bytes(), self.original) + self.assertTrue((backups[0].parent.parent/"report.json").is_file()) + + def test_failure_or_interruption_never_publishes_partial_batch(self): + second = self.workspace.copies_dir/"Copie02.pdf" + second.write_bytes(self.original) + real = crop_margins.process_pdf + for error in (OSError("bad scan"), KeyboardInterrupt()): + with self.subTest(error=type(error).__name__): + def process(source, *args, **kwargs): + if source == second: + raise error + return real(source, *args, **kwargs) + with patch.object(crop_margins, "process_pdf", side_effect=process): + with contextlib.redirect_stdout(io.StringIO()): + with self.assertRaises(type(error)): + crop_margins.run(self.workspace, self.workspace.root, workers=1) + self.assertEqual(self.source.read_bytes(), self.original) + self.assertEqual(second.read_bytes(), self.original) + + def test_existing_coordinates_are_not_silently_invalidated(self): + self.source.with_suffix(".json").write_text('{"list": []}') + with self.assertRaisesRegex(CliError, "coordonnées"): + crop_margins.run(self.workspace, self.workspace.root) + self.assertEqual(self.source.read_bytes(), self.original) + + def test_archiving_removes_crop_backups_but_keeps_processed_pdf_and_logs(self): + with contextlib.redirect_stdout(io.StringIO()): + crop_margins.run(self.workspace, self.workspace.root) + processed = self.source.read_bytes() + self.workspace.correction_file.write_text('{}') + student = self.workspace.return_dir/"Student" + student.mkdir(parents=True) + (student/"answer.jpg").write_bytes(b"return image") + (student/"score.json").write_text('{}') + self.workspace.logs_dir.mkdir(parents=True) + log = self.workspace.logs_dir/"crop_blank_margins.log" + log.write_text("Completed crop") + backups = list(self.workspace.runs_dir.glob("crop-margins-*/Copies/*.pdf")) + reports = list(self.workspace.runs_dir.glob("crop-margins-*/report.json")) + self.assertTrue(backups) + self.assertTrue(reports) + plan = build_cleanup_plan(self.workspace) + for path in backups+reports: + self.assertIn(path, plan.deleted_files) + self.assertIn(self.source, plan.kept_files) + self.assertIn(log, plan.kept_files) + with contextlib.redirect_stdout(io.StringIO()): + apply_cleanup(self.workspace, plan) + self.assertFalse(self.workspace.runs_dir.exists()) + self.assertEqual(self.source.read_bytes(), processed) + self.assertEqual(log.read_text(), "Completed crop") + self.assertTrue((student/"answer.jpg").exists()) + self.assertTrue((student/"score.json").exists()) + + def test_single_copy_selection_cannot_target_original_scans(self): + self.assertEqual(crop_margins.selected_files(self.workspace, self.source), [self.source]) + original = self.workspace.root/"Original.pdf" + original.write_bytes(self.original) + with self.assertRaises(CliError): + crop_margins.selected_files(self.workspace, original) + + def test_parallel_workers_match_serial_results_in_copy_and_page_order(self): + second = self.workspace.copies_dir/"Copie02.pdf" + second.write_bytes(self.original) + serial, parallel = self.workspace.root/"serial", self.workspace.root/"parallel" + serial.mkdir() + parallel.mkdir() + files = [self.source, second] + with contextlib.redirect_stdout(io.StringIO()): + expected = crop_margins.process_copies(files, serial, 1) + actual = crop_margins.process_copies(files, parallel, 2) + self.assertEqual(actual, expected) + self.assertEqual([(r['file'],r['page']) for r in actual], + [(p.name,i) for p in files for i in (1,2)]) + for source in files: + with pymupdf.open(serial/source.name) as a, pymupdf.open(parallel/source.name) as b: + self.assertEqual([list(p.cropbox) for p in a], [list(p.cropbox) for p in b]) + + def test_worker_failure_does_not_replace_any_copy(self): + broken = self.workspace.copies_dir/"Copie02.pdf" + broken.write_bytes(b"not a PDF") + with contextlib.redirect_stdout(io.StringIO()): + with self.assertRaises(Exception): + crop_margins.run(self.workspace, self.workspace.root, workers=2) + self.assertEqual(self.source.read_bytes(), self.original) + self.assertEqual(broken.read_bytes(), b"not a PDF") + self.assertFalse(list(self.workspace.root.glob(".Copies.*.files.tmp"))) + + def test_invalid_worker_count_is_rejected_before_processing(self): + with self.assertRaises(CliError): + crop_margins.run(self.workspace, self.workspace.root, workers=0) + self.assertEqual(self.source.read_bytes(), self.original) + + @unittest.skipIf(os.name == "nt", "SIGINT subprocess check uses Unix signals") + def test_interrupt_stops_parallel_workers_before_removing_staging(self): + import select + with pymupdf.open() as doc: + for _ in range(30): + page = doc.new_page(width=595, height=842) + page.insert_text((100,400), "answer = 42", fontsize=20) + data = doc.tobytes() + self.source.write_bytes(data) + second = self.workspace.copies_dir/"Copie02.pdf" + second.write_bytes(data) + proc = subprocess.Popen( + [sys.executable,"-u","-m","copienator","crop-margins", + str(self.workspace.root),"--workers","2"], + stdout=subprocess.PIPE, stderr=subprocess.STDOUT, text=True, + cwd=Path(__file__).resolve().parents[1]) + try: + while True: + ready, _, _ = select.select([proc.stdout], [], [], 20) + self.assertTrue(ready, "No progress from parallel crop command") + line = proc.stdout.readline() + self.assertTrue(line, "Crop command exited before processing a page") + if "Page " in line: + break + proc.send_signal(signal.SIGINT) + output, _ = proc.communicate(timeout=20) + self.assertEqual(proc.returncode, 130, output) + finally: + if proc.poll() is None: + proc.kill() + proc.communicate() + self.assertEqual(self.source.read_bytes(), data) + self.assertEqual(second.read_bytes(), data) + self.assertFalse(list(self.workspace.root.glob(".Copies.*.files.tmp"))) + + def test_optional_step_sits_between_page_split_and_label_crop(self): + steps = build_workflow(False) + index = next(i for i, step in enumerate(steps) if step.id == "crop_blank_margins") + self.assertEqual(steps[index-1].id, "page_splitter") + self.assertEqual(steps[index+1].id, "cutleft") + self.assertTrue(steps[index].optional) + self.assertFalse(steps[index].auto_start_first_visit) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_gui_convenience.py b/tests/test_gui_convenience.py index 2684eb4..32e6fb0 100644 --- a/tests/test_gui_convenience.py +++ b/tests/test_gui_convenience.py @@ -52,6 +52,22 @@ class GuiConvenienceTests(unittest.TestCase): self.app.open_evaluation_button.invoke() opened.assert_called_once_with(self.evaluation) + def test_optional_blank_crop_can_target_a_copy_or_be_skipped(self): + copies = self.evaluation/"Copies" + copies.mkdir() + source = copies/"Copie01.pdf" + source.touch() + self.app.tree.selection_set("crop_blank_margins") + self.app.update() + self.assertEqual(self.app.current_step.id, "crop_blank_margins") + self.assertEqual(str(self.app.skip_button.cget("state")), "normal") + self.app.copy_var.set(source.name) + self.app._target_selected_copy() + command = self.app._make_command() + self.assertEqual(command[-4:], ["crop-margins", str(source), "--workers", "5"]) + self.app._skip_step() + self.assertEqual(self.app.state_store.step("crop_blank_margins")["status"], "skipped") + def test_redo_targets_selected_copy_and_copies_runnable_command(self): for folder in ("Copies", "Copies Originales"): (self.evaluation / folder).mkdir() diff --git a/tests/test_ink_detection.py b/tests/test_ink_detection.py new file mode 100644 index 0000000..b20e3c7 --- /dev/null +++ b/tests/test_ink_detection.py @@ -0,0 +1,163 @@ +import unittest + +import cv2 +import numpy as np + +from copienator.ink_detection import detect_bounds, _paper_blur + + +class InkDetectionTests(unittest.TestCase): + def test_fast_background_blur_matches_opencv_pixel_for_pixel(self): + rng = np.random.default_rng(42) + for shape in ((97,131), (241,319)): + gray = rng.integers(0,256,shape,dtype=np.uint8) + for dpi in (100,150,200,300): + with self.subTest(shape=shape,dpi=dpi): + expected = cv2.GaussianBlur(gray,(0,0),dpi/8) + np.testing.assert_array_equal(_paper_blur(gray,dpi/8),expected) + + def scan(self): + image = np.full((1200, 850, 3), 255, np.uint8) + for y in range(20, 1200, 20): + cv2.line(image, (0, y), (849, y), (160, 155, 205), 1) + for x in range(10, 850, 20): + cv2.line(image, (x, 0), (x, 1199), (160, 155, 205), 1) + for y in (100, 180, 400, 480, 800, 880, 1050, 1130): + cv2.ellipse(image, (25, y), (12, 20), 0, 0, 300, (155,155,155), 2) + cv2.putText(image, 'x + y = 2', (110, 400), cv2.FONT_HERSHEY_SIMPLEX, + 1, (20, 90, 190), 2) + cv2.putText(image, 'answer = 42', (110, 700), cv2.FONT_HERSHEY_SIMPLEX, + 1, (20, 90, 190), 2) + return image + + def test_crops_past_holes_and_grid(self): + r = detect_bounds(self.scan(), dpi=150) + self.assertGreater(r['top_px'], 250) + self.assertLess(r['top_px'], 370) + self.assertGreater(r['bottom_px'], 700) + self.assertLess(r['bottom_px'], 800) + + def test_isolated_margin_note_survives(self): + image = self.scan() + cv2.putText(image, '1', (4, 1110), cv2.FONT_HERSHEY_SIMPLEX, + .65, (20, 90, 190), 2) + self.assertGreater(detect_bounds(image, dpi=150)['bottom_px'], 1110) + + def test_long_fraction_bar_on_grid_survives(self): + for colour in ((20,90,190), (20,20,20), (190,20,20)): + with self.subTest(colour=colour): + image = self.scan() + cv2.line(image, (100, 1020), (700, 1020), colour, 3) + self.assertGreater(detect_bounds(image, dpi=150)['bottom_px'], 1020) + + def test_black_annotation_survives(self): + image = self.scan() + cv2.putText(image, 'note', (600, 1100), cv2.FONT_HERSHEY_SIMPLEX, + .7, (30,30,30), 2) + self.assertGreater(detect_bounds(image, dpi=150)['bottom_px'], 1100) + + def test_blank_and_pencil_only_pages_are_retained(self): + for pencil in (False, True): + image = np.full((1200,850,3),255,np.uint8) + if pencil: + cv2.putText(image, 'pencil', (100,600), cv2.FONT_HERSHEY_SIMPLEX, + 1, (195,195,195), 2) + r = detect_bounds(image, dpi=150) + self.assertEqual((r['top_px'],r['bottom_px']), (0,1200)) + self.assertEqual(r['status'], 'review-no-ink-seeds') + + def test_skewed_colour_scan(self): + matrix = cv2.getRotationMatrix2D((425,600),2,1) + image = cv2.warpAffine(self.scan(),matrix,(850,1200),borderValue=(255,255,255)) + r = detect_bounds(image,dpi=150) + self.assertGreater(r['top_px'],250) + self.assertLess(r['top_px'],370) + self.assertGreater(r['bottom_px'],715) + self.assertLess(r['bottom_px'],820) + + def test_weak_stroke_attached_to_ink_is_recovered(self): + image = np.full((1200,850,3),255,np.uint8) + cv2.rectangle(image,(400,500),(410,530),(20,90,190),-1) + cv2.rectangle(image,(400,531),(410,540),(130,170,205),-1) + r = detect_bounds(image,dpi=150,padding_mm=0,min_crop_mm=0) + self.assertGreaterEqual(r['bottom_px'],541) + + def dark_grid(self): + image = np.full((1200,850,3),255,np.uint8) + for y in range(20,1200,20): + cv2.line(image,(0,y),(849,y),(65,65,65),1) + for x in range(10,850,20): + cv2.line(image,(x,0),(x,1199),(65,65,65),1) + cv2.putText(image,'x + y = 2',(100,400),cv2.FONT_HERSHEY_SIMPLEX, + 1,(20,20,20),2) + cv2.putText(image,'answer = 42',(100,700),cv2.FONT_HERSHEY_SIMPLEX, + 1,(20,20,20),2) + return image + + def test_dark_grid_does_not_seed_entire_page(self): + r = detect_bounds(self.dark_grid(),dpi=150) + self.assertTrue(r['paper_cleanup']) + self.assertGreater(r['top_px'],250) + self.assertLess(r['top_px'],370) + self.assertGreater(r['bottom_px'],700) + self.assertLess(r['bottom_px'],820) + + def test_faint_isolated_note_on_dark_grid_survives(self): + image = self.dark_grid() + cv2.putText(image,'pencil',(100,1100),cv2.FONT_HERSHEY_SIMPLEX, + .7,(190,190,190),2) + self.assertGreater(detect_bounds(image,dpi=150)['bottom_px'],1100) + + def test_large_unruled_diagram_is_not_paper(self): + image = np.full((1200,850,3),255,np.uint8) + cv2.rectangle(image,(100,100),(750,1100),(20,20,20),3) + r = detect_bounds(image,dpi=150) + self.assertFalse(r['paper_cleanup']) + self.assertLess(r['top_px'],100) + self.assertGreater(r['bottom_px'],1100) + + def test_dark_grid_preserves_black_fraction_bar(self): + image = self.dark_grid() + cv2.line(image,(150,1020),(700,1020),(0,0,0),3) + self.assertGreater(detect_bounds(image,dpi=150)['bottom_px'],1020) + + def test_disconnected_dark_grid_is_still_recognized(self): + image = self.dark_grid() + for x in range(170,850,170): + image[:,x:x+5] = 255 + for y in range(200,1200,200): + image[y:y+5,:] = 255 + cv2.putText(image,'x + y = 2',(100,400),cv2.FONT_HERSHEY_SIMPLEX, + 1,(20,20,20),2) + r = detect_bounds(image,dpi=150) + self.assertTrue(r['paper_cleanup']) + self.assertGreater(r['top_px'],250) + self.assertLess(r['top_px'],370) + self.assertGreater(r['bottom_px'],700) + self.assertLess(r['bottom_px'],850) + + def test_repeated_dark_holes_on_either_side(self): + for right in (False, True): + with self.subTest(right=right): + image = self.dark_grid() + x = 820 if right else 25 + for y in (110, 370, 630, 890, 1130): + cv2.ellipse(image, (x,y), (12,20), 0, 0, 300, (35,35,35), 2) + r = detect_bounds(image,dpi=150) + self.assertLess(r['bottom_px'],850) + # The same column can contain handwriting as well as holes. + cv2.putText(image,'7',(x-5,1060),cv2.FONT_HERSHEY_SIMPLEX, + .7,(20,20,20),2) + r = detect_bounds(image,dpi=150) + self.assertGreater(r['bottom_px'],1060) + + def test_faded_neutral_grid(self): + image = self.dark_grid() + image[np.all(image == 65,axis=2)] = 145 + r = detect_bounds(image,dpi=150) + self.assertTrue(r['paper_cleanup']) + self.assertLess(r['bottom_px'],850) + + +if __name__ == '__main__': + unittest.main()