Standardisation 9
This commit is contained in:
@@ -174,6 +174,7 @@ scripts migrés vers cette convention sont actuellement :
|
|||||||
- =page_splitter.py=, =cutleft.py=, =plotting.py= et
|
- =page_splitter.py=, =cutleft.py=, =plotting.py= et
|
||||||
=splitting_int.py= ;
|
=splitting_int.py= ;
|
||||||
- =gemini_for_labels.py= ;
|
- =gemini_for_labels.py= ;
|
||||||
|
- =gemini_for_enonce.py= et =enonce_info.py= ;
|
||||||
- =correction.py=, =submit_batches.py=, =batch_status.py= et
|
- =correction.py=, =submit_batches.py=, =batch_status.py= et
|
||||||
=fetch_batched_results.py= ;
|
=fetch_batched_results.py= ;
|
||||||
- =annotating.py=, =annotating_with_checks.py= et
|
- =annotating.py=, =annotating_with_checks.py= et
|
||||||
@@ -235,6 +236,12 @@ Dans le dossier de l'évaluation, mettre les fichiers suivants de l'évaluation
|
|||||||
|
|
||||||
- Alternative personnelle : `python enonce_info.py Interro`
|
- Alternative personnelle : `python enonce_info.py Interro`
|
||||||
|
|
||||||
|
Ces deux commandes suivent la convention des scripts standardisés.
|
||||||
|
Leur import ne lance aucun traitement et les erreurs partielles sont
|
||||||
|
distinguées des échecs. Les réponses d'extraction mises en cache par
|
||||||
|
=gemini_for_enonce.py= et le fichier =labels= sont publiés
|
||||||
|
atomiquement.
|
||||||
|
|
||||||
** Prétraitement des copies
|
** Prétraitement des copies
|
||||||
|
|
||||||
Mettre les copies scannées au format pdf dans =Interro=.
|
Mettre les copies scannées au format pdf dans =Interro=.
|
||||||
|
|||||||
+54
-20
@@ -1,16 +1,26 @@
|
|||||||
import sys
|
from __future__ import annotations
|
||||||
import os
|
|
||||||
|
import argparse
|
||||||
import glob
|
import glob
|
||||||
import json
|
import json
|
||||||
import urllib.request
|
import os
|
||||||
import re
|
import re
|
||||||
import subprocess
|
import urllib.request
|
||||||
import tempfile
|
from collections.abc import Sequence
|
||||||
import shutil
|
from uuid import uuid4
|
||||||
|
|
||||||
|
from copienator import (
|
||||||
|
CliError,
|
||||||
|
EvaluationWorkspace,
|
||||||
|
ExitCode,
|
||||||
|
evaluation_parser,
|
||||||
|
execute,
|
||||||
|
workspace_from_args,
|
||||||
|
)
|
||||||
from platform_utils import WindowsLabelError, validate_windows_labels
|
from platform_utils import WindowsLabelError, validate_windows_labels
|
||||||
from utils import compile_to_pdf
|
from utils import compile_to_pdf
|
||||||
|
|
||||||
|
|
||||||
def fetch_and_save_sub_text(ex_id, indices, label, text_path):
|
def fetch_and_save_sub_text(ex_id, indices, label, text_path):
|
||||||
"""Fetches text for a specific sub-question and saves it to Text/{label}.tex"""
|
"""Fetches text for a specific sub-question and saves it to Text/{label}.tex"""
|
||||||
qinds = ",".join(map(str, indices))
|
qinds = ",".join(map(str, indices))
|
||||||
@@ -29,6 +39,7 @@ def fetch_and_save_sub_text(ex_id, indices, label, text_path):
|
|||||||
compile_to_pdf(content, pdf_file)
|
compile_to_pdf(content, pdf_file)
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
print(f"Error fetching sub-text from {url}: {e}")
|
print(f"Error fetching sub-text from {url}: {e}")
|
||||||
|
raise
|
||||||
|
|
||||||
def fetch_and_save_sub_sol(ex_id, indices, label, sol_path):
|
def fetch_and_save_sub_sol(ex_id, indices, label, sol_path):
|
||||||
"""Fetches text for a specific sub-question and saves it to Text/{label}.tex"""
|
"""Fetches text for a specific sub-question and saves it to Text/{label}.tex"""
|
||||||
@@ -48,6 +59,7 @@ def fetch_and_save_sub_sol(ex_id, indices, label, sol_path):
|
|||||||
compile_to_pdf(content, pdf_file)
|
compile_to_pdf(content, pdf_file)
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
print(f"Error fetching sub-text from {url}: {e}")
|
print(f"Error fetching sub-text from {url}: {e}")
|
||||||
|
raise
|
||||||
|
|
||||||
|
|
||||||
ROMANS_CAP = ["", "I", "II", "III", "IV", "V", "VI", "VII", "VIII", "IX", "X"]
|
ROMANS_CAP = ["", "I", "II", "III", "IV", "V", "VI", "VII", "VIII", "IX", "X"]
|
||||||
@@ -127,18 +139,21 @@ def save_split_content(text, path, base_fname, problem):
|
|||||||
f.write(chunk)
|
f.write(chunk)
|
||||||
|
|
||||||
|
|
||||||
def process_directory(directory):
|
def process_directory(workspace: EvaluationWorkspace) -> ExitCode:
|
||||||
|
directory = str(workspace.root)
|
||||||
# Find the first .tex file in the directory
|
# Find the first .tex file in the directory
|
||||||
tex_files = glob.glob(os.path.join(directory, "*.tex"))
|
tex_files = glob.glob(os.path.join(directory, "*.tex"))
|
||||||
if not tex_files:
|
if not tex_files:
|
||||||
print(f"No .tex file found in {directory}. Looking in /Staging/Interro/")
|
print(f"No .tex file found in {directory}. Looking in /Staging/Interro/")
|
||||||
int_name = directory[:-1] if directory.endswith("/") else directory
|
int_name = directory.removesuffix("/")
|
||||||
tex_path = os.path.join(os.path.expanduser("~"), "Prépa/Staging/Interro", f"{int_name}.tex")
|
tex_path = os.path.join(os.path.expanduser("~"), "Prépa/Staging/Interro", f"{int_name}.tex")
|
||||||
if os.path.exists(tex_path):
|
if os.path.exists(tex_path):
|
||||||
tex_file = tex_path
|
tex_file = tex_path
|
||||||
else:
|
else:
|
||||||
print("Not found in ", tex_path)
|
raise CliError(
|
||||||
return
|
f"No .tex input found in {workspace.root}",
|
||||||
|
ExitCode.INVALID_WORKSPACE,
|
||||||
|
)
|
||||||
else:
|
else:
|
||||||
tex_file = tex_files[0]
|
tex_file = tex_files[0]
|
||||||
|
|
||||||
@@ -153,8 +168,10 @@ def process_directory(directory):
|
|||||||
for p in paths.values():
|
for p in paths.values():
|
||||||
os.makedirs(p, exist_ok=True)
|
os.makedirs(p, exist_ok=True)
|
||||||
|
|
||||||
labels_file = os.path.join(directory, "labels")
|
labels_file = workspace.labels_file
|
||||||
|
labels_staging = labels_file.with_name(f".{labels_file.name}.{uuid4().hex}.tmp")
|
||||||
current_ex_num = 1
|
current_ex_num = 1
|
||||||
|
had_errors = False
|
||||||
|
|
||||||
# Read entirely to allow chunking
|
# Read entirely to allow chunking
|
||||||
with open(tex_file, 'r', encoding='utf-8') as f_in:
|
with open(tex_file, 'r', encoding='utf-8') as f_in:
|
||||||
@@ -162,8 +179,11 @@ def process_directory(directory):
|
|||||||
|
|
||||||
# Split by the specific SHEETINFO tag
|
# Split by the specific SHEETINFO tag
|
||||||
blocks = content.split("%%SHEETINFO :")
|
blocks = content.split("%%SHEETINFO :")
|
||||||
|
if len(blocks) == 1:
|
||||||
|
print(f"No SHEETINFO blocks found in {tex_file}")
|
||||||
|
return ExitCode.PARTIAL
|
||||||
|
|
||||||
with open(labels_file, 'w', encoding='utf-8') as f_labels:
|
with open(labels_staging, 'w', encoding='utf-8') as f_labels:
|
||||||
# Skip blocks[0] (content before first SHEETINFO)
|
# Skip blocks[0] (content before first SHEETINFO)
|
||||||
for block in blocks[1:]:
|
for block in blocks[1:]:
|
||||||
parts_line = block.split("\n", 1)
|
parts_line = block.split("\n", 1)
|
||||||
@@ -177,6 +197,7 @@ def process_directory(directory):
|
|||||||
|
|
||||||
try:
|
try:
|
||||||
data = json.loads(json_str)
|
data = json.loads(json_str)
|
||||||
|
block_labels = []
|
||||||
# Construct 'ids' parameter
|
# Construct 'ids' parameter
|
||||||
ex_id = str(data['id'])
|
ex_id = str(data['id'])
|
||||||
selection = data.get('select')
|
selection = data.get('select')
|
||||||
@@ -193,7 +214,7 @@ def process_directory(directory):
|
|||||||
if not indexes:
|
if not indexes:
|
||||||
label = f"Ex {current_ex_num}"
|
label = f"Ex {current_ex_num}"
|
||||||
validate_windows_labels([label])
|
validate_windows_labels([label])
|
||||||
f_labels.write(f"{label}\n")
|
block_labels.append(label)
|
||||||
fetch_and_save_sub_text(ids, [], label, paths['Text2'])
|
fetch_and_save_sub_text(ids, [], label, paths['Text2'])
|
||||||
fetch_and_save_sub_sol(ids, [], label, paths['Sol2'])
|
fetch_and_save_sub_sol(ids, [], label, paths['Sol2'])
|
||||||
else:
|
else:
|
||||||
@@ -201,7 +222,7 @@ def process_directory(directory):
|
|||||||
suffix = format_indices(item['indices'], problem)
|
suffix = format_indices(item['indices'], problem)
|
||||||
label = f"Ex {current_ex_num}" + (f" : {suffix}" if suffix else "")
|
label = f"Ex {current_ex_num}" + (f" : {suffix}" if suffix else "")
|
||||||
validate_windows_labels([label])
|
validate_windows_labels([label])
|
||||||
f_labels.write(f"{label}\n")
|
block_labels.append(label)
|
||||||
fetch_and_save_sub_text(ids, item['indices'], label, paths['Text2'])
|
fetch_and_save_sub_text(ids, item['indices'], label, paths['Text2'])
|
||||||
fetch_and_save_sub_sol(ids, item['indices'], label, paths['Sol2'])
|
fetch_and_save_sub_sol(ids, item['indices'], label, paths['Sol2'])
|
||||||
|
|
||||||
@@ -244,18 +265,31 @@ def process_directory(directory):
|
|||||||
save_split_content(s_text, paths['Sol'], base_filename, problem)
|
save_split_content(s_text, paths['Sol'], base_filename, problem)
|
||||||
save_split_content(p_text, paths['Persp'], base_filename, problem)
|
save_split_content(p_text, paths['Persp'], base_filename, problem)
|
||||||
|
|
||||||
|
for label in block_labels:
|
||||||
|
f_labels.write(f"{label}\n")
|
||||||
current_ex_num += 1
|
current_ex_num += 1
|
||||||
|
|
||||||
except WindowsLabelError:
|
except WindowsLabelError:
|
||||||
|
labels_staging.unlink(missing_ok=True)
|
||||||
raise
|
raise
|
||||||
except json.JSONDecodeError:
|
except json.JSONDecodeError:
|
||||||
print(f"Error decoding JSON in block: {json_str}")
|
print(f"Error decoding JSON in block: {json_str}")
|
||||||
except Exception as e:
|
had_errors = True
|
||||||
|
except Exception as e: # noqa: BLE001 - one malformed exercise is partial
|
||||||
print(f"Error processing block {ex_id if 'ex_id' in locals() else 'unknown'}: {e}")
|
print(f"Error processing block {ex_id if 'ex_id' in locals() else 'unknown'}: {e}")
|
||||||
|
had_errors = True
|
||||||
|
|
||||||
|
labels_staging.replace(labels_file)
|
||||||
|
return ExitCode.PARTIAL if had_errors else ExitCode.SUCCESS
|
||||||
|
|
||||||
|
|
||||||
|
def build_parser() -> argparse.ArgumentParser:
|
||||||
|
return evaluation_parser("Generate statement metadata from SHEETINFO blocks")
|
||||||
|
|
||||||
|
|
||||||
|
def main(argv: Sequence[str] | None = None) -> int:
|
||||||
|
parser = build_parser()
|
||||||
|
return execute(parser, argv, lambda args: process_directory(workspace_from_args(args)))
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
if len(sys.argv) < 2:
|
raise SystemExit(main())
|
||||||
print("Usage: python script.py <Dir>")
|
|
||||||
sys.exit(1)
|
|
||||||
|
|
||||||
process_directory(sys.argv[1])
|
|
||||||
|
|||||||
+91
-51
@@ -1,18 +1,30 @@
|
|||||||
import re
|
from __future__ import annotations
|
||||||
import os
|
|
||||||
import utils
|
|
||||||
import subprocess
|
|
||||||
import sys
|
|
||||||
import argparse
|
import argparse
|
||||||
|
import re
|
||||||
|
from collections.abc import Sequence
|
||||||
|
from concurrent.futures import ThreadPoolExecutor
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from pydantic import BaseModel, Field
|
|
||||||
from typing import List, Union
|
|
||||||
from google import genai
|
from google import genai
|
||||||
from google.genai import types
|
from google.genai import types
|
||||||
from concurrent.futures import ThreadPoolExecutor
|
from pydantic import BaseModel, Field
|
||||||
|
|
||||||
|
import config
|
||||||
|
import utils
|
||||||
|
from copienator import (
|
||||||
|
CliError,
|
||||||
|
EvaluationWorkspace,
|
||||||
|
ExitCode,
|
||||||
|
atomic_write_text,
|
||||||
|
evaluation_parser,
|
||||||
|
execute,
|
||||||
|
workspace_from_args,
|
||||||
|
)
|
||||||
from platform_utils import validate_windows_labels
|
from platform_utils import validate_windows_labels
|
||||||
from utils import compile_to_pdf
|
from utils import compile_to_pdf
|
||||||
|
|
||||||
|
|
||||||
def get_lcp(s1: str, s2: str) -> str:
|
def get_lcp(s1: str, s2: str) -> str:
|
||||||
i = 0
|
i = 0
|
||||||
while i < len(s1) and i < len(s2) and s1[i] == s2[i]:
|
while i < len(s1) and i < len(s2) and s1[i] == s2[i]:
|
||||||
@@ -25,8 +37,6 @@ def get_lcp(s1: str, s2: str) -> str:
|
|||||||
return lcp
|
return lcp
|
||||||
|
|
||||||
|
|
||||||
import config
|
|
||||||
|
|
||||||
MODEL_ID = config.MODEL_LITE_ID
|
MODEL_ID = config.MODEL_LITE_ID
|
||||||
api_key = config.API_KEY
|
api_key = config.API_KEY
|
||||||
|
|
||||||
@@ -36,7 +46,7 @@ class QuestionOnlyItem(BaseModel):
|
|||||||
question_content: str = Field(description="The source text of the question, strictly extracted from the enonce file, EXCLUDING the label itself.")
|
question_content: str = Field(description="The source text of the question, strictly extracted from the enonce file, EXCLUDING the label itself.")
|
||||||
|
|
||||||
class ExamQuestions(BaseModel):
|
class ExamQuestions(BaseModel):
|
||||||
questions: List[QuestionOnlyItem]
|
questions: list[QuestionOnlyItem]
|
||||||
|
|
||||||
# --- Modèles pour la Requête 2 ---
|
# --- Modèles pour la Requête 2 ---
|
||||||
class SolutionOnlyItem(BaseModel):
|
class SolutionOnlyItem(BaseModel):
|
||||||
@@ -44,7 +54,7 @@ class SolutionOnlyItem(BaseModel):
|
|||||||
solution_content: str = Field(description="The source text of the solution, strictly extracted from the correction file.")
|
solution_content: str = Field(description="The source text of the solution, strictly extracted from the correction file.")
|
||||||
|
|
||||||
class ExamSolutions(BaseModel):
|
class ExamSolutions(BaseModel):
|
||||||
solutions: List[SolutionOnlyItem]
|
solutions: list[SolutionOnlyItem]
|
||||||
|
|
||||||
# --- Modèles pour la Requête 3 ---
|
# --- Modèles pour la Requête 3 ---
|
||||||
class ExtractedContext(BaseModel):
|
class ExtractedContext(BaseModel):
|
||||||
@@ -53,7 +63,7 @@ class ExtractedContext(BaseModel):
|
|||||||
context_content: str = Field(description="The source text of the definitions, notations, or hypotheses, extracted from the enonce.")
|
context_content: str = Field(description="The source text of the definitions, notations, or hypotheses, extracted from the enonce.")
|
||||||
|
|
||||||
class ExamContext(BaseModel):
|
class ExamContext(BaseModel):
|
||||||
contexts: List[ExtractedContext]
|
contexts: list[ExtractedContext]
|
||||||
|
|
||||||
# --- Modèles pour la Requête 4 (Barèmes) ---
|
# --- Modèles pour la Requête 4 (Barèmes) ---
|
||||||
class RubricItem(BaseModel):
|
class RubricItem(BaseModel):
|
||||||
@@ -61,7 +71,7 @@ class RubricItem(BaseModel):
|
|||||||
rubric_content: str = Field(description="Le barème détaillé en français.")
|
rubric_content: str = Field(description="Le barème détaillé en français.")
|
||||||
|
|
||||||
class GroupRubrics(BaseModel):
|
class GroupRubrics(BaseModel):
|
||||||
rubrics: List[RubricItem]
|
rubrics: list[RubricItem]
|
||||||
|
|
||||||
PROMPT_4 = """Je te fournis les questions, le contexte éventuel, et les corrections pour un groupe de questions d'un examen.
|
PROMPT_4 = """Je te fournis les questions, le contexte éventuel, et les corrections pour un groupe de questions d'un examen.
|
||||||
Ta tâche :
|
Ta tâche :
|
||||||
@@ -87,10 +97,10 @@ class ContextItem(BaseModel):
|
|||||||
content: str # Juste une string encapsulée pour le différencier facilement
|
content: str # Juste une string encapsulée pour le différencier facilement
|
||||||
|
|
||||||
class ExamExtraction(BaseModel):
|
class ExamExtraction(BaseModel):
|
||||||
items: List[Union[QuestionItem, ContextItem]] # Liste mixte
|
items: list[QuestionItem | ContextItem] # Liste mixte
|
||||||
|
|
||||||
class GroupedExamExtraction(BaseModel):
|
class GroupedExamExtraction(BaseModel):
|
||||||
groups: List[List[Union[QuestionItem, ContextItem]]]
|
groups: list[list[QuestionItem | ContextItem]]
|
||||||
|
|
||||||
PROMPT_1 = """I am providing:
|
PROMPT_1 = """I am providing:
|
||||||
1. A PDF of an exam (`enonce.pdf`)
|
1. A PDF of an exam (`enonce.pdf`)
|
||||||
@@ -138,15 +148,20 @@ the `Let N, M be two commutating matrices` part is not a question itself, and is
|
|||||||
Return the result as a JSON list.
|
Return the result as a JSON list.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
def find_file(folder: Path, base_name: str) -> Path:
|
def find_file(folder: Path, base_name: str) -> Path | None:
|
||||||
for ext in [".org", ".tex"]:
|
for ext in [".org", ".tex"]:
|
||||||
path = folder / f"{base_name}{ext}"
|
path = folder / f"{base_name}{ext}"
|
||||||
if path.is_file():
|
if path.is_file():
|
||||||
return path
|
return path
|
||||||
return None
|
return None
|
||||||
|
|
||||||
def process_exam(folder_path: str, restart: bool = False):
|
def process_exam(
|
||||||
folder = Path(folder_path)
|
workspace: EvaluationWorkspace,
|
||||||
|
restart: bool = False,
|
||||||
|
*,
|
||||||
|
api_client=None,
|
||||||
|
) -> ExitCode:
|
||||||
|
folder = workspace.root
|
||||||
|
|
||||||
cache_dir = folder / "Cache"
|
cache_dir = folder / "Cache"
|
||||||
tmp_dir = folder / "Tmp"
|
tmp_dir = folder / "Tmp"
|
||||||
@@ -168,15 +183,21 @@ def process_exam(folder_path: str, restart: bool = False):
|
|||||||
if not correction_path: missing.append("correction.org or correction.tex")
|
if not correction_path: missing.append("correction.org or correction.tex")
|
||||||
|
|
||||||
if missing:
|
if missing:
|
||||||
print(f"Error: Missing files in {folder}: {', '.join(missing)}")
|
raise CliError(
|
||||||
sys.exit(1)
|
f"Missing files in {folder}: {', '.join(missing)}",
|
||||||
|
ExitCode.INVALID_WORKSPACE,
|
||||||
|
)
|
||||||
|
|
||||||
print("Reading files...")
|
print("Reading files...")
|
||||||
pdf_bytes = pdf_path.read_bytes()
|
pdf_bytes = pdf_path.read_bytes()
|
||||||
enonce_text = enonce_path.read_text(encoding="utf-8")
|
enonce_text = enonce_path.read_text(encoding="utf-8")
|
||||||
correction_text = correction_path.read_text(encoding="utf-8")
|
correction_text = correction_path.read_text(encoding="utf-8")
|
||||||
|
|
||||||
client = genai.Client(api_key=api_key)
|
if api_client is None:
|
||||||
|
if not api_key:
|
||||||
|
raise CliError("GEMINI_API_KEY is not configured")
|
||||||
|
api_client = genai.Client(api_key=api_key)
|
||||||
|
client = api_client
|
||||||
|
|
||||||
# ==========================================
|
# ==========================================
|
||||||
# REQUÊTE 1 : Extraction des Énoncés
|
# REQUÊTE 1 : Extraction des Énoncés
|
||||||
@@ -210,7 +231,7 @@ def process_exam(folder_path: str, restart: bool = False):
|
|||||||
)
|
)
|
||||||
response_q_text = response_q.text
|
response_q_text = response_q.text
|
||||||
print("Saving questions to cache...")
|
print("Saving questions to cache...")
|
||||||
cache_q_file.write_text(response_q_text, encoding="utf-8")
|
atomic_write_text(cache_q_file, response_q_text)
|
||||||
|
|
||||||
questions_data = ExamQuestions.model_validate_json(response_q_text)
|
questions_data = ExamQuestions.model_validate_json(response_q_text)
|
||||||
|
|
||||||
@@ -248,7 +269,7 @@ def process_exam(folder_path: str, restart: bool = False):
|
|||||||
)
|
)
|
||||||
response_s_text = response_s.text
|
response_s_text = response_s.text
|
||||||
print("Saving solutions to cache...")
|
print("Saving solutions to cache...")
|
||||||
cache_s_file.write_text(response_s_text, encoding="utf-8")
|
atomic_write_text(cache_s_file, response_s_text)
|
||||||
|
|
||||||
solutions_data = ExamSolutions.model_validate_json(response_s_text)
|
solutions_data = ExamSolutions.model_validate_json(response_s_text)
|
||||||
|
|
||||||
@@ -284,7 +305,7 @@ def process_exam(folder_path: str, restart: bool = False):
|
|||||||
)
|
)
|
||||||
response_c_text = response_c.text
|
response_c_text = response_c.text
|
||||||
print("Saving context to cache...")
|
print("Saving context to cache...")
|
||||||
cache_c_file.write_text(response_c_text, encoding="utf-8")
|
atomic_write_text(cache_c_file, response_c_text)
|
||||||
|
|
||||||
context_data = ExamContext.model_validate_json(response_c_text)
|
context_data = ExamContext.model_validate_json(response_c_text)
|
||||||
|
|
||||||
@@ -388,7 +409,6 @@ def process_exam(folder_path: str, restart: bool = False):
|
|||||||
for g_indices in q_group_indices:
|
for g_indices in q_group_indices:
|
||||||
group_items = []
|
group_items = []
|
||||||
first_q_idx = g_indices[0]
|
first_q_idx = g_indices[0]
|
||||||
last_q_idx = g_indices[-1]
|
|
||||||
|
|
||||||
for q_idx in g_indices:
|
for q_idx in g_indices:
|
||||||
q_item = questions_only[q_idx]
|
q_item = questions_only[q_idx]
|
||||||
@@ -497,14 +517,13 @@ def process_exam(folder_path: str, restart: bool = False):
|
|||||||
|
|
||||||
# 2. Actual Parsing
|
# 2. Actual Parsing
|
||||||
grouped_items = []
|
grouped_items = []
|
||||||
current_raw_group = [] # Stores (is_context, label_or_flag, content)
|
|
||||||
all_new_q_labels = []
|
all_new_q_labels = []
|
||||||
|
|
||||||
# Pass 1: Read all edited lines and collect question labels in sequence
|
# Pass 1: Read all edited lines and collect question labels in sequence
|
||||||
for line in edited_lines:
|
for line in edited_lines:
|
||||||
if line == "---" or " ### " not in line:
|
if line == "---" or " ### " not in line:
|
||||||
continue
|
continue
|
||||||
lbl, content_raw = line.split(" ### ", 1)
|
lbl, _content_raw = line.split(" ### ", 1)
|
||||||
lbl = lbl.strip()
|
lbl = lbl.strip()
|
||||||
if lbl != "CONTEXT":
|
if lbl != "CONTEXT":
|
||||||
all_new_q_labels.append(lbl)
|
all_new_q_labels.append(lbl)
|
||||||
@@ -606,10 +625,6 @@ def process_exam(folder_path: str, restart: bool = False):
|
|||||||
validate_windows_labels(labels_list)
|
validate_windows_labels(labels_list)
|
||||||
|
|
||||||
# Save labels and proceed
|
# Save labels and proceed
|
||||||
with open(folder / "labels", 'w', encoding='utf-8') as f_labels:
|
|
||||||
for label in labels_list:
|
|
||||||
f_labels.write(f"{label}\n")
|
|
||||||
|
|
||||||
grouped_extraction = GroupedExamExtraction(groups=grouped_items)
|
grouped_extraction = GroupedExamExtraction(groups=grouped_items)
|
||||||
|
|
||||||
# 2. Setup output directories
|
# 2. Setup output directories
|
||||||
@@ -628,8 +643,7 @@ def process_exam(folder_path: str, restart: bool = False):
|
|||||||
).strip().lower()
|
).strip().lower()
|
||||||
|
|
||||||
if answer not in ("y", "yes"):
|
if answer not in ("y", "yes"):
|
||||||
print("Aborted.")
|
raise CliError("Output replacement aborted", ExitCode.INVALID_ARGUMENTS)
|
||||||
sys.exit(1)
|
|
||||||
# Empty each directory
|
# Empty each directory
|
||||||
for d in dirs:
|
for d in dirs:
|
||||||
if d.exists():
|
if d.exists():
|
||||||
@@ -645,6 +659,7 @@ def process_exam(folder_path: str, restart: bool = False):
|
|||||||
|
|
||||||
|
|
||||||
print("Writing grouped question and solution files...")
|
print("Writing grouped question and solution files...")
|
||||||
|
processing_errors = []
|
||||||
|
|
||||||
for group in grouped_extraction.groups:
|
for group in grouped_extraction.groups:
|
||||||
q_items = [item for item in group if isinstance(item, QuestionItem)]
|
q_items = [item for item in group if isinstance(item, QuestionItem)]
|
||||||
@@ -690,8 +705,9 @@ def process_exam(folder_path: str, restart: bool = False):
|
|||||||
)
|
)
|
||||||
rubrics_data = GroupRubrics.model_validate_json(response_r.text)
|
rubrics_data = GroupRubrics.model_validate_json(response_r.text)
|
||||||
rubrics_map = {r.label: r.rubric_content for r in rubrics_data.rubrics}
|
rubrics_map = {r.label: r.rubric_content for r in rubrics_data.rubrics}
|
||||||
except Exception as e:
|
except Exception as e: # noqa: BLE001 - remote API boundary
|
||||||
print(f"Error generating rubric for group {labels[0]}: {e}")
|
print(f"Error generating rubric for group {labels[0]}: {e}")
|
||||||
|
processing_errors.append(str(e))
|
||||||
rubrics_map = {}
|
rubrics_map = {}
|
||||||
|
|
||||||
# 1. Compute the common prefix for the group
|
# 1. Compute the common prefix for the group
|
||||||
@@ -737,7 +753,7 @@ def process_exam(folder_path: str, restart: bool = False):
|
|||||||
elif isinstance(item, ContextItem):
|
elif isinstance(item, ContextItem):
|
||||||
raw_ctx = item.content.strip()
|
raw_ctx = item.content.strip()
|
||||||
tabulated_ctx = "\t" + re.sub(r'\n\s*', '\n\t', raw_ctx)
|
tabulated_ctx = "\t" + re.sub(r'\n\s*', '\n\t', raw_ctx)
|
||||||
text_content_lines.append(f"CONTEXT :")
|
text_content_lines.append("CONTEXT :")
|
||||||
text_content_lines.append(tabulated_ctx)
|
text_content_lines.append(tabulated_ctx)
|
||||||
|
|
||||||
# --- Save context to Text2 (Concatenating if exists) ---
|
# --- Save context to Text2 (Concatenating if exists) ---
|
||||||
@@ -762,29 +778,53 @@ def process_exam(folder_path: str, restart: bool = False):
|
|||||||
# ==========================================
|
# ==========================================
|
||||||
all_tex_files = list(text2_dir.glob("*.tex")) + list(sol2_dir.glob("*.tex"))
|
all_tex_files = list(text2_dir.glob("*.tex")) + list(sol2_dir.glob("*.tex"))
|
||||||
|
|
||||||
def compile_worker(tex_file: Path):
|
def compile_worker(tex_file: Path) -> str | None:
|
||||||
"""Helper to read content and call the utility function."""
|
"""Helper to read content and call the utility function."""
|
||||||
try:
|
try:
|
||||||
content = tex_file.read_text(encoding="utf-8")
|
content = tex_file.read_text(encoding="utf-8")
|
||||||
pdf_path = tex_file.with_suffix(".pdf")
|
pdf_path = tex_file.with_suffix(".pdf")
|
||||||
compile_to_pdf(content, pdf_path)
|
compile_to_pdf(content, pdf_path)
|
||||||
except Exception as e:
|
except Exception as e: # noqa: BLE001 - compiler worker boundary
|
||||||
print(f"Error compiling {tex_file.name}: {e}")
|
return f"Error compiling {tex_file.name}: {e}"
|
||||||
|
return None
|
||||||
|
|
||||||
print(f"Compiling {len(all_tex_files)} files to PDF using 4 threads...")
|
print(f"Compiling {len(all_tex_files)} files to PDF using 4 threads...")
|
||||||
with ThreadPoolExecutor(max_workers=4) as executor:
|
with ThreadPoolExecutor(max_workers=4) as executor:
|
||||||
executor.map(compile_worker, all_tex_files)
|
compile_errors = [
|
||||||
|
error for error in executor.map(compile_worker, all_tex_files) if error
|
||||||
|
]
|
||||||
|
for error in compile_errors:
|
||||||
|
print(error)
|
||||||
|
processing_errors.extend(compile_errors)
|
||||||
|
atomic_write_text(
|
||||||
|
workspace.labels_file,
|
||||||
|
"".join(f"{label}\n" for label in labels_list),
|
||||||
|
)
|
||||||
|
|
||||||
|
return ExitCode.PARTIAL if processing_errors else ExitCode.SUCCESS
|
||||||
|
|
||||||
|
|
||||||
|
def build_parser() -> argparse.ArgumentParser:
|
||||||
|
parser = evaluation_parser("Extract exam and solution code via Gemini")
|
||||||
|
parser.add_argument(
|
||||||
|
"--restart",
|
||||||
|
action="store_true",
|
||||||
|
help="Ignore cached Gemini extraction responses",
|
||||||
|
)
|
||||||
|
return parser
|
||||||
|
|
||||||
|
|
||||||
|
def main(argv: Sequence[str] | None = None) -> int:
|
||||||
|
parser = build_parser()
|
||||||
|
return execute(
|
||||||
|
parser,
|
||||||
|
argv,
|
||||||
|
lambda args: process_exam(
|
||||||
|
workspace_from_args(args),
|
||||||
|
restart=args.restart,
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
if not api_key:
|
raise SystemExit(main())
|
||||||
print("Error: GEMINI_API_KEY environment variable is not set.")
|
|
||||||
sys.exit(1)
|
|
||||||
|
|
||||||
parser = argparse.ArgumentParser(description="Extract exam and solution code via Gemini.")
|
|
||||||
parser.add_argument("folder", help="Directory containing the exam files")
|
|
||||||
parser.add_argument("--restart", action="store_true", help="Ignore cache files and re-run extraction requests.")
|
|
||||||
|
|
||||||
|
|
||||||
args = parser.parse_args()
|
|
||||||
process_exam(args.folder, restart=args.restart)
|
|
||||||
|
|||||||
@@ -322,6 +322,10 @@ class StandardCliTests(unittest.TestCase):
|
|||||||
"gemini_for_labels": load_script_module(
|
"gemini_for_labels": load_script_module(
|
||||||
"gemini_for_labels.py", "gemini_for_labels"
|
"gemini_for_labels.py", "gemini_for_labels"
|
||||||
),
|
),
|
||||||
|
"gemini_for_enonce": load_script_module(
|
||||||
|
"gemini_for_enonce.py", "gemini_for_enonce"
|
||||||
|
),
|
||||||
|
"enonce_info": load_script_module("enonce_info.py", "enonce_info"),
|
||||||
"correction": load_script_module("correction.py", "correction"),
|
"correction": load_script_module("correction.py", "correction"),
|
||||||
"submit_batches": load_script_module(
|
"submit_batches": load_script_module(
|
||||||
"submit_batches.py", "submit_batches"
|
"submit_batches.py", "submit_batches"
|
||||||
@@ -372,6 +376,8 @@ class StandardCliTests(unittest.TestCase):
|
|||||||
"page_splitter": [missing],
|
"page_splitter": [missing],
|
||||||
"plotting": [missing],
|
"plotting": [missing],
|
||||||
"gemini_for_labels": [missing],
|
"gemini_for_labels": [missing],
|
||||||
|
"gemini_for_enonce": [missing],
|
||||||
|
"enonce_info": [missing],
|
||||||
"correction": [missing],
|
"correction": [missing],
|
||||||
"submit_batches": [missing],
|
"submit_batches": [missing],
|
||||||
"fetch_batched_results": [missing],
|
"fetch_batched_results": [missing],
|
||||||
@@ -419,6 +425,16 @@ class StandardCliTests(unittest.TestCase):
|
|||||||
steps = {step.id: step for step in build_workflow(True)}
|
steps = {step.id: step for step in build_workflow(True)}
|
||||||
evaluation = "Evaluation with spaces"
|
evaluation = "Evaluation with spaces"
|
||||||
cases = {
|
cases = {
|
||||||
|
"gemini_for_enonce": (
|
||||||
|
"statement",
|
||||||
|
"gemini",
|
||||||
|
{"target": evaluation, "restart": True},
|
||||||
|
),
|
||||||
|
"enonce_info": (
|
||||||
|
"statement",
|
||||||
|
"personal",
|
||||||
|
{"target": evaluation},
|
||||||
|
),
|
||||||
"export": ("export", "default", {"target": evaluation, "refaire": True}),
|
"export": ("export", "default", {"target": evaluation, "refaire": True}),
|
||||||
"import": ("import", "default", {"target": evaluation, "refaire": True}),
|
"import": ("import", "default", {"target": evaluation, "refaire": True}),
|
||||||
"giving_names": (
|
"giving_names": (
|
||||||
@@ -534,6 +550,21 @@ class StandardCliTests(unittest.TestCase):
|
|||||||
self.assertEqual(parsed.operation, step_id)
|
self.assertEqual(parsed.operation, step_id)
|
||||||
self.assertEqual(str(parsed.evaluation), evaluation)
|
self.assertEqual(str(parsed.evaluation), evaluation)
|
||||||
|
|
||||||
|
def test_enonce_info_preserves_labels_when_no_blocks_are_found(self) -> None:
|
||||||
|
module = self.modules["enonce_info"]
|
||||||
|
with tempfile.TemporaryDirectory() as directory:
|
||||||
|
evaluation = Path(directory) / "Exam"
|
||||||
|
evaluation.mkdir()
|
||||||
|
(evaluation / "source.tex").write_text(
|
||||||
|
"No SHEETINFO blocks\n", encoding="utf-8"
|
||||||
|
)
|
||||||
|
(evaluation / "labels").write_text("Existing\n", encoding="utf-8")
|
||||||
|
self.assertEqual(
|
||||||
|
module.process_directory(EvaluationWorkspace(evaluation)), 4
|
||||||
|
)
|
||||||
|
self.assertEqual((evaluation / "labels").read_text(), "Existing\n")
|
||||||
|
self.assertFalse(list(evaluation.glob(".labels.*.tmp")))
|
||||||
|
|
||||||
def test_export_main_copies_outputs(self) -> None:
|
def test_export_main_copies_outputs(self) -> None:
|
||||||
module = self.modules["export"]
|
module = self.modules["export"]
|
||||||
with tempfile.TemporaryDirectory() as directory:
|
with tempfile.TemporaryDirectory() as directory:
|
||||||
|
|||||||
Reference in New Issue
Block a user