564 lines
19 KiB
Python
564 lines
19 KiB
Python
from __future__ import annotations
|
|
|
|
import argparse
|
|
import io
|
|
import os
|
|
import re
|
|
import subprocess
|
|
import tempfile
|
|
from collections.abc import Sequence
|
|
from pathlib import Path
|
|
|
|
import matplotlib
|
|
import matplotlib.colors as mcolors
|
|
import PIL.ImageOps
|
|
|
|
matplotlib.use("Agg")
|
|
|
|
from matplotlib import pyplot as plt
|
|
from pdf2image import convert_from_path
|
|
from PIL import Image, ImageDraw, ImageFont
|
|
|
|
from copienator import utils
|
|
from copienator.configuration import LATEX_ANOT_AFTER, LATEX_ANOT_BEFORE
|
|
from copienator import (
|
|
EvaluationWorkspace,
|
|
ExitCode,
|
|
atomic_write_json,
|
|
evaluation_parser,
|
|
execute,
|
|
workspace_from_args,
|
|
)
|
|
from copienator.annotation_data import load_annotation_data
|
|
from copienator.answer_info import build_answer_info
|
|
from copienator.filesystem import staged_directory
|
|
from copienator.feedback_boxes import valid_feedback_box
|
|
from copienator.utils import natural_key
|
|
|
|
MARGIN_LEFT = 300
|
|
ANNOT_WIDTH = 600
|
|
|
|
|
|
|
|
def make_dictionary(root_dir, refaire=False, refaire_list=None):
|
|
"""Compatibility wrapper used by the annotation-reading scripts."""
|
|
workspace = EvaluationWorkspace(Path(root_dir))
|
|
loaded = load_annotation_data(
|
|
workspace,
|
|
refaire_list=(refaire_list or []) if refaire else None,
|
|
)
|
|
return loaded.data
|
|
|
|
|
|
def make_base_image(pdf_path):
|
|
pages = convert_from_path(pdf_path)
|
|
|
|
# Calculate total dimensions
|
|
total_h = sum(page.height for page in pages)
|
|
max_w = max(page.width for page in pages)
|
|
|
|
# Create concatenated base image
|
|
base_img = Image.new("RGBA", (max_w, total_h), "white")
|
|
|
|
current_y = 0
|
|
for page in pages:
|
|
base_img.paste(page.convert("RGBA"), (0, current_y))
|
|
current_y += page.height
|
|
return (base_img, total_h, max_w)
|
|
|
|
def normalize_mathtext(text):
|
|
"""
|
|
Replaces LaTeX shortcuts not supported by Matplotlib's mathtext parser.
|
|
e.g. \\le -> \\leq, \\ge -> \\geq
|
|
Using lookahead (?![a-zA-Z]) prevents replacing \\left with \\leqft.
|
|
"""
|
|
text = re.sub(r'\\le(?![a-zA-Z])', r'\\leq', text)
|
|
text = re.sub(r'\\ge(?![a-zA-Z])', r'\\geq', text)
|
|
text = re.sub(r'\\implies', r'\\Rightarrow', text)
|
|
# Sometimes, Gemini escapes too much ? Not sure
|
|
text = text.replace("\\\\", "\\")
|
|
text = text.replace("\\llbracket", "[\\![")
|
|
text = text.replace("\\rrbracket", "]\\!]")
|
|
text = text.replace("\\R", "\\mathbb{R}")
|
|
text = text.replace("\\N", "\\mathbb{N}")
|
|
text = text.replace("\\Z", "\\mathbb{Z}")
|
|
text = text.replace("\\C", "\\mathbb{C}")
|
|
text = text.replace("\\Q", "\\mathbb{Q}")
|
|
# Sometimes, Gemini doesn't escape enough. In the json, you should have \\f
|
|
text = text.replace('\f', r'\f')
|
|
text = re.sub('\u0010', "", text)
|
|
return text
|
|
|
|
def wrap_latex_text(text, width_chars):
|
|
"""
|
|
Wraps text but keeps LaTeX math blocks ($...$) intact.
|
|
"""
|
|
# 1. Split text into chunks of: text, math, text, math...
|
|
# The regex looks for $...$ (non-greedy).
|
|
parts = re.split(r'(\$[^\$]+\$)', text)
|
|
|
|
# 2. Tokenize: Break plain text by spaces, keep math blocks whole.
|
|
tokens = []
|
|
for part in parts:
|
|
if part.startswith('$') and part.endswith('$'):
|
|
tokens.append(part) # Keep math block distinct
|
|
else:
|
|
tokens.extend(part.split()) # Split normal text by whitespace
|
|
|
|
# 3. Reconstruct lines using textwrap logic
|
|
lines = []
|
|
current_line = []
|
|
current_length = 0
|
|
|
|
for token in tokens:
|
|
# +1 for the space we will add
|
|
token_len = len(token)
|
|
|
|
if current_length + token_len + 1 > width_chars:
|
|
lines.append(" ".join(current_line))
|
|
current_line = [token]
|
|
current_length = token_len
|
|
else:
|
|
current_line.append(token)
|
|
current_length += token_len + 1
|
|
|
|
if current_line:
|
|
lines.append(" ".join(current_line))
|
|
|
|
res = "\n".join(lines)
|
|
return res
|
|
|
|
def render_latex_text(text, width_px, bg_color=(255, 255, 255, 255), max_lines=None,
|
|
fontsize=14):
|
|
# 1. Fix unsupported symbols
|
|
text = normalize_mathtext(text)
|
|
|
|
dpi = 100
|
|
fig_width = width_px / dpi
|
|
|
|
# Estimate characters per line based on width and font size (heuristic)
|
|
# FontSize 12 approx 0.5 inches wide for ~15 chars usually,
|
|
# but let's approximate: Width (inches) * ~10 chars/inch for size 12
|
|
chars_per_line = int(fig_width * 10)
|
|
|
|
# Pre-wrap the text respecting LaTeX boundaries
|
|
wrapped_text = wrap_latex_text(text, chars_per_line)
|
|
|
|
# Dynamic height based on actual number of lines
|
|
num_lines = wrapped_text.count('\n') + 1
|
|
if max_lines and num_lines > max_lines:
|
|
# logic to truncate if strictly necessary, or just expand
|
|
pass
|
|
|
|
# 0.3 inches per line buffer
|
|
fig_height = num_lines * 0.3 + 0.2
|
|
|
|
fig = plt.figure(figsize=(fig_width, fig_height), dpi=dpi)
|
|
|
|
# NOTE: wrap=False because we did it ourselves
|
|
plt.text(0.01, 0.95, wrapped_text, fontsize=fontsize,
|
|
verticalalignment='top', horizontalalignment='left',
|
|
wrap=False)
|
|
|
|
plt.axis('off')
|
|
|
|
buf = io.BytesIO()
|
|
plt.savefig(buf, format='png', bbox_inches='tight', pad_inches=0.1, transparent=True)
|
|
plt.close(fig)
|
|
buf.seek(0)
|
|
|
|
img = Image.open(buf).convert("RGBA")
|
|
|
|
# Create background
|
|
final_img = Image.new("RGBA", img.size, bg_color)
|
|
final_img.alpha_composite(img)
|
|
return final_img
|
|
|
|
def render_real_latex_text(text, width_px, bg_color=(255, 255, 255, 255), max_lines=None, fontsize=19):
|
|
dpi = 100
|
|
width_in = width_px / dpi
|
|
line_spacing = int(fontsize * 1.2)
|
|
|
|
# Use the 'standalone' class with 'varwidth' to auto-crop height while restricting width
|
|
header = LATEX_ANOT_BEFORE.substitute(
|
|
width_in=width_in, fontsize=fontsize, line_spacing=line_spacing
|
|
)
|
|
latex_template = f"{header}{text}{LATEX_ANOT_AFTER}"
|
|
|
|
with tempfile.TemporaryDirectory() as temp_dir:
|
|
tex_path = os.path.join(temp_dir, 'text.tex')
|
|
pdf_path = os.path.join(temp_dir, 'text.pdf')
|
|
|
|
with open(tex_path, 'w', encoding='utf-8') as f:
|
|
f.write(latex_template)
|
|
|
|
# Compile to PDF
|
|
subprocess.run(
|
|
['pdflatex', '-interaction=nonstopmode', 'text.tex'],
|
|
cwd=temp_dir,
|
|
stdout=subprocess.DEVNULL,
|
|
stderr=subprocess.DEVNULL,
|
|
check=False,
|
|
)
|
|
|
|
if not os.path.exists(pdf_path):
|
|
raise RuntimeError("LaTeX compilation failed. Check your LaTeX syntax.")
|
|
|
|
# Convert PDF to grayscale (ignoring pdf2image's broken transparency)
|
|
images = convert_from_path(pdf_path, dpi=dpi)
|
|
gray_img = images[0].convert("L")
|
|
|
|
# 1. Invert grayscale to create an alpha mask (white bg = 0, black text = 255)
|
|
alpha_mask = PIL.ImageOps.invert(gray_img)
|
|
|
|
# 2. Create a transparent image with black text using the mask
|
|
text_img = Image.new("RGBA", gray_img.size, (0, 0, 0, 255))
|
|
text_img.putalpha(alpha_mask)
|
|
|
|
# 3. Create the requested background and composite the text over it
|
|
final_img = Image.new("RGBA", text_img.size, bg_color)
|
|
final_img.alpha_composite(text_img)
|
|
|
|
# (Optional) Truncate image height if max_lines is strictly enforced
|
|
if max_lines:
|
|
max_height_px = int((fontsize * 1.2 / 72.0) * dpi * max_lines) # Points to pixels
|
|
if final_img.height > max_height_px:
|
|
final_img = final_img.crop((0, 0, final_img.width, max_height_px))
|
|
|
|
return final_img
|
|
|
|
def color(score):
|
|
t = max(0.0, min(1.0, float(score) / 4.0))
|
|
t = t*1.5 - 0.25
|
|
t = max(0.0, min(1.0, t))
|
|
red = 200 * (1 - t)
|
|
green = 150 * t
|
|
return mcolors.to_hex((red/255, green/255, 0))
|
|
|
|
def render_score_text(label, score, error, width_px, fontsize=30,
|
|
bg_color=(255, 255, 255, 255),
|
|
with_error=True, id=None):
|
|
|
|
# 1. Build text segments: (text, color, is_bold)
|
|
parts = []
|
|
default_color = (0, 0, 0, 255)
|
|
|
|
prefix = f"{id} " if id else ""
|
|
prefix += f"{label} Note : "
|
|
parts.append((prefix, default_color, False))
|
|
|
|
parts.append((str(score), color(score), True))
|
|
|
|
if error and error != "null" and with_error:
|
|
fontsize=18
|
|
parts.append((" ", default_color, False))
|
|
parts.append((str(error), "orange", True))
|
|
|
|
# 2. Setup Image
|
|
height_px = 80 # roughly matches fig_height=0.8 at 100 dpi
|
|
img = Image.new("RGBA", (int(width_px), height_px), bg_color)
|
|
draw = ImageDraw.Draw(img)
|
|
|
|
# 3. Load Fonts
|
|
try:
|
|
font_regular = ImageFont.truetype("DejaVuSans.ttf", fontsize)
|
|
font_bold = ImageFont.truetype("DejaVuSans-Bold.ttf", fontsize)
|
|
except OSError:
|
|
# Fallback for systems without specific TTFs readily available
|
|
print("here")
|
|
try:
|
|
font_regular = ImageFont.load_default(size=fontsize) # Pillow >= 10.1.0
|
|
except TypeError:
|
|
print("there")
|
|
font_regular = ImageFont.load_default()
|
|
font_bold = font_regular
|
|
|
|
# 4. Draw segments horizontally
|
|
x, y = int(width_px * 0.125), int(height_px * 0.2)
|
|
|
|
for text, text_color, is_bold in parts:
|
|
f = font_bold if is_bold else font_regular
|
|
draw.text((x, y), text, fill=text_color, font=f)
|
|
|
|
# Advance X position by the width of the drawn text
|
|
bbox = draw.textbbox((x, y), text, font=f)
|
|
x = bbox[2]
|
|
|
|
return img
|
|
|
|
A4_WIDTH_200DPI = 1654
|
|
TARGET_MIN_WIDTH = int(A4_WIDTH_200DPI * 0.9) # 1406 pixels
|
|
def compose_label_image(base_img, label, result, hmin,
|
|
render_fn=render_real_latex_text,
|
|
draw_callback=None,
|
|
with_error=True,
|
|
with_empty=False,
|
|
more_right=False,
|
|
with_id=None):
|
|
"""
|
|
Composes the final image with annotations.
|
|
|
|
Args:
|
|
base_img: The source PDF converted to image.
|
|
label: Label name (e.g. "Ex1").
|
|
result: The JSON result object (score, feedbacks).
|
|
hmin: Vertical offset coordinate.
|
|
render_fn: Function to render text to image (allows threading injection).
|
|
draw_callback: Optional function(type, draw_obj, position_dict, data_dict)
|
|
called when elements are placed. Used for checkboxes.
|
|
"""
|
|
|
|
left_pad = 0
|
|
if base_img.width < TARGET_MIN_WIDTH:
|
|
total_missing = TARGET_MIN_WIDTH - base_img.width
|
|
left_pad = min(total_missing, MARGIN_LEFT)
|
|
|
|
new_base = Image.new("RGB", (TARGET_MIN_WIDTH, base_img.height), "white")
|
|
new_base.paste(base_img, (left_pad, 0))
|
|
base_img = new_base
|
|
|
|
score = result.get('score', 0)
|
|
error = result.get('error', "")
|
|
feedbacks = result.get('feedback', [])
|
|
|
|
if error == "empty-answer" and not with_empty:
|
|
return None, 0
|
|
|
|
|
|
# Filter deleted items (used by reading_annotations.py)
|
|
feedbacks = [f for f in feedbacks if "to_delete" not in f]
|
|
# Never guess where an invalid rectangle belongs, or lose its comment.
|
|
# Use a copy so rendering cannot mutate saved correction data.
|
|
normalized = []
|
|
for feedback in feedbacks:
|
|
box = feedback.get("box_2d")
|
|
if box is not None and not valid_feedback_box(box):
|
|
print(f"Warning: Copie{with_id or ''} {label}: invalid feedback box {box!r}; displaying the comment without a rectangle.")
|
|
feedback = {**feedback, "box_2d": None}
|
|
normalized.append(feedback)
|
|
feedbacks = normalized
|
|
|
|
global_fb = [f for f in feedbacks if not f.get('box_2d')]
|
|
local_fb = [f for f in feedbacks if f.get('box_2d')]
|
|
local_fb.sort(key=lambda x: x['box_2d'][0])
|
|
|
|
# 1. Prepare Headers
|
|
header_elements = []
|
|
|
|
if more_right:
|
|
width = base_img.width // 2
|
|
else:
|
|
width = base_img.width // 2 - 150
|
|
img_score = render_score_text(label, score, error, width, with_error=with_error,
|
|
id=with_id)
|
|
header_elements.append({"type": "score", "img": img_score, "data": result})
|
|
|
|
# Global Feedbacks
|
|
for idx, fb in enumerate(global_fb):
|
|
img_fb = render_fn(fb['text'], base_img.width)
|
|
header_elements.append({"type": "global_fb", "img": img_fb, "data": fb, "index": idx})
|
|
|
|
# Calculate Header Height
|
|
header_height = sum(el["img"].height for el in header_elements)
|
|
total_height = base_img.height + header_height
|
|
|
|
# Create Canvas
|
|
final_img = Image.new("RGB", (base_img.width + MARGIN_LEFT, total_height), "white")
|
|
|
|
# Draw Headers
|
|
current_y = 0
|
|
draw = ImageDraw.Draw(final_img, "RGBA")
|
|
|
|
for el in header_elements:
|
|
if el["type"] == "score" and more_right:
|
|
final_img.paste(el["img"], (150, current_y))
|
|
else:
|
|
final_img.paste(el["img"], (0, current_y))
|
|
|
|
|
|
if draw_callback:
|
|
# Hook for checkboxes
|
|
draw_callback("header_item", draw,
|
|
{"x": 0, "y": current_y, "w": el["img"].width, "h": el["img"].height},
|
|
el)
|
|
current_y += el["img"].height
|
|
|
|
# Paste Base Image
|
|
image_offset_y = current_y
|
|
final_img.paste(base_img, (MARGIN_LEFT, image_offset_y))
|
|
|
|
# 2. Draw Local Annotations
|
|
draw = ImageDraw.Draw(final_img, "RGBA") # Refresh draw object
|
|
last_text_bottom = 0
|
|
|
|
for idx, fb in enumerate(local_fb):
|
|
box = fb.get('box_2d')
|
|
ymin, xmin, ymax, xmax = box
|
|
|
|
target_ymin = (ymin - hmin) + image_offset_y
|
|
target_ymax = (ymax - hmin) + image_offset_y
|
|
target_xmin = xmin + MARGIN_LEFT + left_pad
|
|
target_xmax = xmax + MARGIN_LEFT + left_pad
|
|
|
|
# Draw Rectangle (if not suppressed)
|
|
if "norectangle" not in fb:
|
|
draw.rectangle([target_xmin, target_ymin, target_xmax, target_ymax], outline="red", width=3)
|
|
|
|
if draw_callback:
|
|
draw_callback("local_rect", draw,
|
|
{"box": [target_xmin, target_ymin, target_xmax, target_ymax]},
|
|
{"data": fb, "index": idx})
|
|
|
|
# Render Text
|
|
txt_img = render_fn(fb['text'], width_px=ANNOT_WIDTH,
|
|
bg_color=(255, 200, 200, 180), max_lines=None)
|
|
|
|
# Calculate Position
|
|
center_y = (target_ymin + target_ymax) / 2
|
|
paste_y = center_y - (txt_img.height / 2)
|
|
paste_y = max(paste_y, image_offset_y)
|
|
|
|
if paste_y < last_text_bottom:
|
|
paste_y = last_text_bottom + 5
|
|
|
|
# Resize canvas if needed
|
|
required_height = int(paste_y + txt_img.height + 20)
|
|
if required_height > final_img.height:
|
|
new_final = Image.new("RGB", (final_img.width, required_height), "white")
|
|
new_final.paste(final_img, (0, 0))
|
|
final_img = new_final
|
|
draw = ImageDraw.Draw(final_img, "RGBA")
|
|
|
|
# Paste Text
|
|
final_img.paste(txt_img, (10, int(paste_y)), mask=txt_img)
|
|
|
|
if draw_callback:
|
|
draw_callback("local_text", draw,
|
|
{"x": 10, "y": int(paste_y), "w": txt_img.width, "h": txt_img.height},
|
|
{"data": fb, "index": idx})
|
|
|
|
last_text_bottom = paste_y + txt_img.height
|
|
|
|
return final_img, header_height
|
|
|
|
def process_student(student_id, labels_data, root_dir, all_labels, overwrite):
|
|
"""Helper function to process a single student."""
|
|
|
|
# Prepare output directory: Dir/Anot_CopieID
|
|
output_dir = Path(root_dir) / "Anot" / f"Copie{student_id}"
|
|
|
|
# Check if already processed (Concat.jpg exists)
|
|
concat_path = output_dir / "Concat.jpg"
|
|
if concat_path.exists() and not overwrite:
|
|
print(f"Skipping Copie {student_id} (Concat.jpg exists)")
|
|
return "skipped"
|
|
|
|
print("Processing :", student_id)
|
|
problems = False
|
|
with staged_directory(output_dir) as staging:
|
|
d_notes = dict.fromkeys(all_labels, "")
|
|
label_images = []
|
|
answer_labels = []
|
|
sorted_labels = sorted(labels_data.items(), key=lambda item: natural_key(item[0]))
|
|
|
|
for label, content in sorted_labels:
|
|
pdf_full_path = content.get('pdf_path')
|
|
if not pdf_full_path or not Path(pdf_full_path).exists():
|
|
print(f"File not found: {pdf_full_path}")
|
|
problems = True
|
|
continue
|
|
try:
|
|
base_img, _, _ = make_base_image(pdf_full_path)
|
|
except Exception as exc: # noqa: BLE001 - PDF/LaTeX backends vary
|
|
print(f"Error converting {pdf_full_path}: {exc}")
|
|
problems = True
|
|
continue
|
|
|
|
result = content.get('result', {})
|
|
coordinates = content.get('coordinates', (0, 0))
|
|
d_notes[label] = str(result.get('score', 0))
|
|
final_img, _ = compose_label_image(
|
|
base_img,
|
|
label,
|
|
result,
|
|
coordinates[0],
|
|
with_empty=True,
|
|
render_fn=render_real_latex_text,
|
|
)
|
|
final_img.save(staging / f"{label}.jpg")
|
|
if result.get('error', "") != "empty-answer":
|
|
label_images.append(final_img)
|
|
answer_labels.append(label)
|
|
|
|
atomic_write_json(staging / "score.json", d_notes)
|
|
atomic_write_json(staging / "info.json", build_answer_info(
|
|
d_notes, labels_data, answer_labels
|
|
))
|
|
if label_images:
|
|
max_w = max(image.width for image in label_images)
|
|
total_h = sum(image.height for image in label_images)
|
|
canvas = Image.new('RGB', (max_w, total_h))
|
|
current_y = 0
|
|
for image in label_images:
|
|
canvas.paste(image, (0, current_y))
|
|
current_y += image.height
|
|
canvas.save(staging / "Concat.jpg")
|
|
elif labels_data:
|
|
problems = True
|
|
return "partial" if problems else "success"
|
|
|
|
|
|
def process_correction(root_dir, data, all_labels, overwrite=False):
|
|
# Ne pas thread cette application
|
|
# 1. Il faut protéger les appels à matplotlib
|
|
# 2. tu vas perdre les erreurs
|
|
statuses = [
|
|
process_student(student_id, labels, root_dir, all_labels, overwrite)
|
|
for student_id, labels in sorted(data.items())
|
|
]
|
|
return ExitCode.PARTIAL if "partial" in statuses else ExitCode.SUCCESS
|
|
|
|
|
|
def build_parser() -> argparse.ArgumentParser:
|
|
parser = evaluation_parser("Generate simple annotated copies.")
|
|
parser.add_argument(
|
|
"--overwrite",
|
|
action="store_true",
|
|
help="Replace existing student output directories",
|
|
)
|
|
return parser
|
|
|
|
|
|
def run(workspace: EvaluationWorkspace, *, overwrite: bool = False) -> ExitCode:
|
|
workspace.require_files("labels", "correction.json")
|
|
workspace.require_directories("Copies", "Par label")
|
|
labels = utils.read_all_labels(workspace.root)
|
|
loaded = load_annotation_data(workspace)
|
|
for warning in loaded.warnings:
|
|
print(f"Warning: {warning}")
|
|
if not loaded.data:
|
|
print("Warning: no annotation data was found.")
|
|
return ExitCode.PARTIAL
|
|
result = process_correction(
|
|
workspace.root,
|
|
loaded.data,
|
|
labels,
|
|
overwrite=overwrite,
|
|
)
|
|
if loaded.warnings and result == ExitCode.SUCCESS:
|
|
return ExitCode.PARTIAL
|
|
return result
|
|
|
|
|
|
def main(argv: Sequence[str] | None = None) -> int:
|
|
parser = build_parser()
|
|
return execute(
|
|
parser,
|
|
argv,
|
|
lambda args: run(workspace_from_args(args), overwrite=args.overwrite),
|
|
)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
raise SystemExit(main())
|