From 90112125e2785e6c6e569ea2517fa5a5a8e12e03 Mon Sep 17 00:00:00 2001 From: lduignan Date: Sat, 14 Feb 2026 19:58:23 +0100 Subject: [PATCH 01/11] add exo7.py task file --- community_tasks/exo7.py | 167 ++++++++++++++++++++++++++++++++++++++++ 1 file changed, 167 insertions(+) create mode 100644 community_tasks/exo7.py diff --git a/community_tasks/exo7.py b/community_tasks/exo7.py new file mode 100644 index 000000000..07d8b9bf0 --- /dev/null +++ b/community_tasks/exo7.py @@ -0,0 +1,167 @@ +""" +Custom evaluation task for the Exo7 French math multiple-choice benchmark. + +This module implements a multi-label MC task with TruthfulQA MC2-style +probability scoring: log-likelihoods are normalized to probabilities, +and accuracy is measured as the probability mass assigned to correct answers. + +Dataset: Lduignan1/exo7_mc +""" + +import numpy as np + +from lighteval.metrics.metrics_sample import SampleLevelComputation +from lighteval.metrics.utils.metric_utils import SampleLevelMetric +from lighteval.models.model_output import ModelResponse +from lighteval.tasks.default_prompts import LETTER_INDICES +from lighteval.tasks.lighteval_task import LightevalTaskConfig +from lighteval.tasks.requests import Doc, SamplingMethod + + +# --- Custom MC2 metric --- + + +class Exo7MCMetric(SampleLevelComputation): + """MC2-style probability mass metric for multi-label multiple choice. + + Converts log-likelihoods to probabilities, normalizes them, and returns + the total probability mass on the correct answers. + """ + + def compute(self, model_response: ModelResponse, doc: Doc, **kwargs): + logprobs = model_response.logprobs + probs = np.exp(np.array(logprobs)) + probs_norm = probs / np.sum(probs) + + labels = np.array(doc.specific["labels"]) + return float(np.sum(probs_norm[labels == 1])) + + +exo7_mc_metric = SampleLevelMetric( + metric_name="acc", + sample_level_fn=Exo7MCMetric(), + category=SamplingMethod.LOGPROBS, + corpus_level_fn=np.mean, + higher_is_better=True, +) + + +# --- Prompt functions --- + +FEWSHOT_EXAMPLES = """\ +Pour les questions suivantes, il peut y avoir plusieurs bonnes réponses. Choisissez une seule bonne réponse possible. + +Question : +Soient $f,g$ deux fonctions définies sur $\\Rr$. Quelles sont les assertions vraies ? + +A. $f - 2g$ est une fonction définie sur $\\Rr$ +B. $f^2 \\times g$ est une fonction définie sur $\\Rr$ +C. $\\frac{f}{g^2}$ est une fonction définie sur $\\Rr$ +D. $\\sqrt{f+g}$ est une fonction définie sur $\\Rr$ + +Réponse : $f^2 \\times g$ est une fonction définie sur $\\Rr$ + + +Question : +Etant donné que $\\displaystyle f(3)=1$ et $f'(3)=5$. Une équation de la tangente à $\\mathcal{C}_f$ au point $(3,1)$ est : + +A. $y=1(x-3)+5=x+2$ +B. $y=1(x-3)-5=x-8$ +C. $y=5(x-3)-1=5x-16$ +D. $y=5(x-3)+1=5x-14$ + +Réponse : $y=5(x-3)+1=5x-14$ + + +Question : +On considère les deux applications suivantes : +$$\\begin{array}{rccc}f:&\\Rr&\\to&\\Rr\\\\ +& x&\\to & \\sin x \\end{array} \\quad \\mbox{et} \\quad \\begin{array}{rccc}g:&\\Rr^2&\\to&\\Rr^2\\\\ +& (x,y)&\\to &(y,x). \\end{array}$$ + +Quelles sont les assertions vraies ? + +A. $f(0)=0$ +B. $f$ est une application linéaire +C. $g(x,y)=g(y,x)$, pour tout $(x,y) \\in \\Rr^2$ +D. $g$ est une application linéaire + +Réponse : $f(0)=0$ + +""" + +ZEROSHOT_INSTRUCTION = "Pour la question suivante, il peut y avoir plusieurs bonnes réponses. Choisissez une seule bonne réponse possible.\n\n" + + +def _build_query(line): + """Build the question + lettered choices portion of the prompt.""" + choices = line["targets"]["choices"] + query = f"Question :\n{line['question']}\n\n" + query += "".join( + f"{letter}. {choice}\n" + for letter, choice in zip(LETTER_INDICES, choices) + ) + query += "\nRéponse :" + return query + + +def _build_doc(line, query, task_name): + """Build a Doc from the formatted query and dataset line.""" + choices = line["targets"]["choices"] + labels = line["targets"]["labels"] + gold_index = [i for i, label in enumerate(labels) if label == 1] + + return Doc( + task_name=task_name, + query=query, + choices=choices, + gold_index=gold_index, + specific={"labels": labels}, + ) + + +def prompt_exo7_mc_zeroshot(line, task_name: str = None): + query = ZEROSHOT_INSTRUCTION + _build_query(line) + return _build_doc(line, query, task_name) + + +def prompt_exo7_mc_fewshot(line, task_name: str = None): + query = FEWSHOT_EXAMPLES + _build_query(line) + return _build_doc(line, query, task_name) + + +# --- Task configs --- + +exo7_mc_zeroshot_task = LightevalTaskConfig( + name="exo7:mc_zeroshot", + prompt_function=prompt_exo7_mc_zeroshot, + suite=["community"], + hf_repo="Lduignan1/exo7_mc", + hf_subset="default", + hf_avail_splits=["test"], + evaluation_splits=["test"], + few_shots_split=None, + few_shots_select=None, + generation_size=1, + metrics=[exo7_mc_metric], + stop_sequence=["\n"], + version=0, +) + +exo7_mc_fewshot_task = LightevalTaskConfig( + name="exo7:mc_fewshot", + prompt_function=prompt_exo7_mc_fewshot, + suite=["community"], + hf_repo="Lduignan1/exo7_mc", + hf_subset="default", + hf_avail_splits=["test"], + evaluation_splits=["test"], + few_shots_split=None, + few_shots_select=None, + generation_size=1, + metrics=[exo7_mc_metric], + stop_sequence=["\n"], + version=0, +) + +TASKS_TABLE = [exo7_mc_zeroshot_task, exo7_mc_fewshot_task] From c071f8cea1d0926c980e873d84c63aade31664bc Mon Sep 17 00:00:00 2001 From: lduignan Date: Sun, 15 Feb 2026 18:39:05 +0100 Subject: [PATCH 02/11] update hf_repo for exo7 --- community_tasks/exo7.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/community_tasks/exo7.py b/community_tasks/exo7.py index 07d8b9bf0..263efbd81 100644 --- a/community_tasks/exo7.py +++ b/community_tasks/exo7.py @@ -136,7 +136,7 @@ def prompt_exo7_mc_fewshot(line, task_name: str = None): name="exo7:mc_zeroshot", prompt_function=prompt_exo7_mc_zeroshot, suite=["community"], - hf_repo="Lduignan1/exo7_mc", + hf_repo="Lduignan1/Exo7", hf_subset="default", hf_avail_splits=["test"], evaluation_splits=["test"], @@ -152,7 +152,7 @@ def prompt_exo7_mc_fewshot(line, task_name: str = None): name="exo7:mc_fewshot", prompt_function=prompt_exo7_mc_fewshot, suite=["community"], - hf_repo="Lduignan1/exo7_mc", + hf_repo="Lduignan1/Exo7", hf_subset="default", hf_avail_splits=["test"], evaluation_splits=["test"], From 0467774373b8c6b05e59c4efc36255834c5ff607 Mon Sep 17 00:00:00 2001 From: lduignan Date: Tue, 21 Apr 2026 02:22:44 +0200 Subject: [PATCH 03/11] Enhance Exo7 task with detailed dataset description and implement generative metrics for multi-label evaluation --- community_tasks/exo7.py | 283 ++++++++++++++++++++++++++-------------- 1 file changed, 185 insertions(+), 98 deletions(-) diff --git a/community_tasks/exo7.py b/community_tasks/exo7.py index 263efbd81..f0c57d785 100644 --- a/community_tasks/exo7.py +++ b/community_tasks/exo7.py @@ -1,13 +1,29 @@ """ -Custom evaluation task for the Exo7 French math multiple-choice benchmark. +name: +Exo7 -This module implements a multi-label MC task with TruthfulQA MC2-style -probability scoring: log-likelihoods are normalized to probabilities, -and accuracy is measured as the probability mass assigned to correct answers. +dataset: +OpenLLM-BPI/Exo7MCQ + +abstract: +Exo7 is a dataset of multi-label multiple-choice math questions for French undergraduate +students, sourced from http://exo7.emath.fr/. Many items have more than one correct answer. +Two scoring paths are exposed, both zero-shot: a logprob path (MCF, Hybrid) using a +TruthfulQA MC2-style probability-mass metric, and a generative path that asks the model to +emit "Réponse : A, C" and scores with set-F1 and exact-set-match. + +languages: +french + +tags: +math, question-answering, multiple-choice, multi-label + +paper: -Dataset: Lduignan1/exo7_mc """ +import re + import numpy as np from lighteval.metrics.metrics_sample import SampleLevelComputation @@ -16,13 +32,19 @@ from lighteval.tasks.default_prompts import LETTER_INDICES from lighteval.tasks.lighteval_task import LightevalTaskConfig from lighteval.tasks.requests import Doc, SamplingMethod +from lighteval.tasks.templates.multichoice import get_mcq_prompt_function +from lighteval.tasks.templates.utils.formulation import ( + HybridFormulation, + MCFFormulation, +) +from lighteval.utils.language import Language -# --- Custom MC2 metric --- +# --- Custom logprob mass metric --- class Exo7MCMetric(SampleLevelComputation): - """MC2-style probability mass metric for multi-label multiple choice. + """Probability mass metric for multi-label multiple choice. Converts log-likelihoods to probabilities, normalizes them, and returns the total probability mass on the correct answers. @@ -46,122 +68,187 @@ def compute(self, model_response: ModelResponse, doc: Doc, **kwargs): ) -# --- Prompt functions --- +# --- Generative metrics (multi-letter answer) --- -FEWSHOT_EXAMPLES = """\ -Pour les questions suivantes, il peut y avoir plusieurs bonnes réponses. Choisissez une seule bonne réponse possible. -Question : -Soient $f,g$ deux fonctions définies sur $\\Rr$. Quelles sont les assertions vraies ? +_RESPONSE_RE = re.compile(r"(?:^|\n)\s*[Rr][ée]ponse\s*:?\s*([^\n]*)") +_LETTER_RE = re.compile(r"\b[A-Z]\b") -A. $f - 2g$ est une fonction définie sur $\\Rr$ -B. $f^2 \\times g$ est une fonction définie sur $\\Rr$ -C. $\\frac{f}{g^2}$ est une fonction définie sur $\\Rr$ -D. $\\sqrt{f+g}$ est une fonction définie sur $\\Rr$ -Réponse : $f^2 \\times g$ est une fonction définie sur $\\Rr$ +def _extract_letters(text: str, valid: set) -> set: + """Extract the set of answer letters from a generative response. + Prefers the last line starting with "Réponse :" (the instructed format); + otherwise falls back to the last non-empty line. Keeps only letters in + the valid set for this question. Uses word boundaries so isolated + capitals (e.g. "A, C") match but letters inside words ("Aucune", "Vrai") + do not. + """ + if not text: + return set() + matches = list(_RESPONSE_RE.finditer(text)) + if matches: + target = matches[-1].group(1) + else: + lines = [line for line in text.strip().splitlines() if line.strip()] + target = lines[-1] if lines else "" + return {c for c in _LETTER_RE.findall(target) if c in valid} -Question : -Etant donné que $\\displaystyle f(3)=1$ et $f'(3)=5$. Une équation de la tangente à $\\mathcal{C}_f$ au point $(3,1)$ est : -A. $y=1(x-3)+5=x+2$ -B. $y=1(x-3)-5=x-8$ -C. $y=5(x-3)-1=5x-16$ -D. $y=5(x-3)+1=5x-14$ +class Exo7GenerativeF1(SampleLevelComputation): + """Set-F1 between predicted and gold letter sets.""" -Réponse : $y=5(x-3)+1=5x-14$ + def compute(self, model_response: ModelResponse, doc: Doc, **kwargs): + pred_text = model_response.text[0] if model_response.text else "" + valid = set(doc.choices) + gold = set(doc.specific["correct_letters"]) + pred = _extract_letters(pred_text, valid) + if not gold and not pred: + return 1.0 + if not gold or not pred: + return 0.0 + tp = len(pred & gold) + if tp == 0: + return 0.0 + precision = tp / len(pred) + recall = tp / len(gold) + return 2 * precision * recall / (precision + recall) + + +class Exo7GenerativeExactMatch(SampleLevelComputation): + """1.0 iff the predicted letter set exactly matches the gold set.""" + def compute(self, model_response: ModelResponse, doc: Doc, **kwargs): + pred_text = model_response.text[0] if model_response.text else "" + valid = set(doc.choices) + gold = set(doc.specific["correct_letters"]) + pred = _extract_letters(pred_text, valid) + return float(pred == gold) -Question : -On considère les deux applications suivantes : -$$\\begin{array}{rccc}f:&\\Rr&\\to&\\Rr\\\\ -& x&\\to & \\sin x \\end{array} \\quad \\mbox{et} \\quad \\begin{array}{rccc}g:&\\Rr^2&\\to&\\Rr^2\\\\ -& (x,y)&\\to &(y,x). \\end{array}$$ -Quelles sont les assertions vraies ? +exo7_generative_f1_metric = SampleLevelMetric( + metric_name="f1", + sample_level_fn=Exo7GenerativeF1(), + category=SamplingMethod.GENERATIVE, + corpus_level_fn=np.mean, + higher_is_better=True, +) -A. $f(0)=0$ -B. $f$ est une application linéaire -C. $g(x,y)=g(y,x)$, pour tout $(x,y) \\in \\Rr^2$ -D. $g$ est une application linéaire +exo7_generative_exact_metric = SampleLevelMetric( + metric_name="exact_match", + sample_level_fn=Exo7GenerativeExactMatch(), + category=SamplingMethod.GENERATIVE, + corpus_level_fn=np.mean, + higher_is_better=True, +) -Réponse : $f(0)=0$ -""" +# --- Prompt function --- -ZEROSHOT_INSTRUCTION = "Pour la question suivante, il peut y avoir plusieurs bonnes réponses. Choisissez une seule bonne réponse possible.\n\n" +INSTRUCTION = ( + "Pour la question suivante, une ou plusieurs propositions peuvent être correctes. " + "Évaluez chaque proposition." +) -def _build_query(line): - """Build the question + lettered choices portion of the prompt.""" - choices = line["targets"]["choices"] - query = f"Question :\n{line['question']}\n\n" - query += "".join( - f"{letter}. {choice}\n" - for letter, choice in zip(LETTER_INDICES, choices) - ) - query += "\nRéponse :" - return query - - -def _build_doc(line, query, task_name): - """Build a Doc from the formatted query and dataset line.""" - choices = line["targets"]["choices"] - labels = line["targets"]["labels"] - gold_index = [i for i, label in enumerate(labels) if label == 1] - - return Doc( - task_name=task_name, - query=query, - choices=choices, - gold_index=gold_index, - specific={"labels": labels}, +def _make_prompt_fn(formulation): + base_fn = get_mcq_prompt_function( + Language.FRENCH, + lambda line: { + "question": line["question"], + "choices": line["targets"]["choices"], + "gold_idx": [i for i, label in enumerate(line["targets"]["labels"]) if label == 1], + "instruction": INSTRUCTION, + }, + formulation=formulation, ) + def prompt_fn(line, task_name: str = None): + doc = base_fn(line, task_name) + doc.specific = {"labels": line["targets"]["labels"]} + return doc + + return prompt_fn + + +GENERATIVE_INSTRUCTION_TEMPLATE = ( + "Pour la question suivante, une ou plusieurs propositions peuvent être correctes. " + "Évaluez chaque proposition, puis indiquez toutes les lettres des propositions correctes. " + "La dernière ligne de votre réponse doit être au format suivant : " + "'Réponse : $LETTRES' (sans les guillemets) où $LETTRES est une liste de lettres parmi " + "{valid_letters} séparées par des virgules (par exemple 'Réponse : A, C'). " + "Réfléchissez étape par étape avant de répondre." +) + -def prompt_exo7_mc_zeroshot(line, task_name: str = None): - query = ZEROSHOT_INSTRUCTION + _build_query(line) - return _build_doc(line, query, task_name) +def _make_generative_prompt_fn(): + def prompt_fn(line, task_name: str = None): + choices = line["targets"]["choices"] + labels = line["targets"]["labels"] + letters = list(LETTER_INDICES[: len(choices)]) + correct_letters = [letters[i] for i, label in enumerate(labels) if label == 1] + instruction = GENERATIVE_INSTRUCTION_TEMPLATE.format(valid_letters=", ".join(letters)) + choices_str = "\n".join( + f"{letter}) {choice.strip()}" for letter, choice in zip(letters, choices) + ) + query = f"{instruction}\n\n{line['question'].strip()}\n\n{choices_str}" -def prompt_exo7_mc_fewshot(line, task_name: str = None): - query = FEWSHOT_EXAMPLES + _build_query(line) - return _build_doc(line, query, task_name) + doc = Doc( + task_name=task_name, + query=query, + choices=letters, + gold_index=[i for i, label in enumerate(labels) if label == 1], + instruction=instruction, + ) + doc.specific = { + "correct_letters": correct_letters, + "labels": labels, + } + return doc + + return prompt_fn # --- Task configs --- -exo7_mc_zeroshot_task = LightevalTaskConfig( - name="exo7:mc_zeroshot", - prompt_function=prompt_exo7_mc_zeroshot, - suite=["community"], - hf_repo="Lduignan1/Exo7", - hf_subset="default", - hf_avail_splits=["test"], - evaluation_splits=["test"], - few_shots_split=None, - few_shots_select=None, - generation_size=1, - metrics=[exo7_mc_metric], - stop_sequence=["\n"], - version=0, -) +FORMULATIONS = [MCFFormulation(), HybridFormulation()] + + +def _make_task(formulation): + return LightevalTaskConfig( + name=f"exo7_{formulation.name.lower()}", + prompt_function=_make_prompt_fn(formulation), + suite=["community"], + hf_repo="OpenLLM-BPI/Exo7MCQ", + hf_subset="default", + hf_avail_splits=["test"], + evaluation_splits=["test"], + few_shots_split=None, + few_shots_select=None, + generation_size=1, + metrics=[exo7_mc_metric], + stop_sequence=["\n"], + version=0, + ) + + +def _make_generative_task(): + return LightevalTaskConfig( + name="exo7_generative", + prompt_function=_make_generative_prompt_fn(), + suite=["community"], + hf_repo="OpenLLM-BPI/Exo7MCQ", + hf_subset="default", + hf_avail_splits=["test"], + evaluation_splits=["test"], + few_shots_split=None, + few_shots_select=None, + generation_size=16384, + metrics=[exo7_generative_f1_metric, exo7_generative_exact_metric], + stop_sequence=[], + version=0, + ) -exo7_mc_fewshot_task = LightevalTaskConfig( - name="exo7:mc_fewshot", - prompt_function=prompt_exo7_mc_fewshot, - suite=["community"], - hf_repo="Lduignan1/Exo7", - hf_subset="default", - hf_avail_splits=["test"], - evaluation_splits=["test"], - few_shots_split=None, - few_shots_select=None, - generation_size=1, - metrics=[exo7_mc_metric], - stop_sequence=["\n"], - version=0, -) -TASKS_TABLE = [exo7_mc_zeroshot_task, exo7_mc_fewshot_task] +TASKS_TABLE = [_make_task(formulation) for formulation in FORMULATIONS] + [_make_generative_task()] From fcd1519c68b6a4303adb5170b89fc01360adc431 Mon Sep 17 00:00:00 2001 From: lduignan Date: Wed, 29 Apr 2026 16:59:13 +0200 Subject: [PATCH 04/11] Refactor Exo7MCMetric to length-normalize logprobs and improve stability in probability calculations --- community_tasks/exo7.py | 16 ++++++++++++++-- 1 file changed, 14 insertions(+), 2 deletions(-) diff --git a/community_tasks/exo7.py b/community_tasks/exo7.py index f0c57d785..170d1ac0b 100644 --- a/community_tasks/exo7.py +++ b/community_tasks/exo7.py @@ -51,8 +51,20 @@ class Exo7MCMetric(SampleLevelComputation): """ def compute(self, model_response: ModelResponse, doc: Doc, **kwargs): - logprobs = model_response.logprobs - probs = np.exp(np.array(logprobs)) + logprobs = np.array(model_response.logprobs) + + # Length-normalize per choice. Prefer per-choice token counts from the + # model wrapper; fall back to character lengths of the formulation + # targets if the backend didn't populate output_tokens. + if model_response.output_tokens: + lengths = np.array([max(len(t), 1) for t in model_response.output_tokens]) + else: + lengths = np.array([max(len(c), 1) for c in doc.choices]) + + norm_logprobs = logprobs / lengths + + # Stability shift + softmax + probs = np.exp(norm_logprobs - np.max(norm_logprobs)) probs_norm = probs / np.sum(probs) labels = np.array(doc.specific["labels"]) From d53184caf735917eb11a79a720f5c71822d5841a Mon Sep 17 00:00:00 2001 From: lduignan Date: Tue, 5 May 2026 13:30:46 +0200 Subject: [PATCH 05/11] add fallback to handle response with answer in \boxed --- community_tasks/exo7.py | 20 +++++++++++++------- 1 file changed, 13 insertions(+), 7 deletions(-) diff --git a/community_tasks/exo7.py b/community_tasks/exo7.py index 170d1ac0b..c96b134bf 100644 --- a/community_tasks/exo7.py +++ b/community_tasks/exo7.py @@ -84,6 +84,7 @@ def compute(self, model_response: ModelResponse, doc: Doc, **kwargs): _RESPONSE_RE = re.compile(r"(?:^|\n)\s*[Rr][ée]ponse\s*:?\s*([^\n]*)") +_BOXED_RE = re.compile(r"\\boxed\s*\{([^}]*)\}") _LETTER_RE = re.compile(r"\b[A-Z]\b") @@ -91,10 +92,11 @@ def _extract_letters(text: str, valid: set) -> set: """Extract the set of answer letters from a generative response. Prefers the last line starting with "Réponse :" (the instructed format); - otherwise falls back to the last non-empty line. Keeps only letters in - the valid set for this question. Uses word boundaries so isolated - capitals (e.g. "A, C") match but letters inside words ("Aucune", "Vrai") - do not. + failing that, the contents of the last ``\\boxed{...}`` (math-tuned + models like Qwen2.5-Math default to this); otherwise the last non-empty + line. Keeps only letters in the valid set. Uses word boundaries so + isolated capitals (e.g. "A, C") match but letters inside words + ("Aucune", "Vrai") do not. """ if not text: return set() @@ -102,8 +104,12 @@ def _extract_letters(text: str, valid: set) -> set: if matches: target = matches[-1].group(1) else: - lines = [line for line in text.strip().splitlines() if line.strip()] - target = lines[-1] if lines else "" + boxed = list(_BOXED_RE.finditer(text)) + if boxed: + target = boxed[-1].group(1) + else: + lines = [line for line in text.strip().splitlines() if line.strip()] + target = lines[-1] if lines else "" return {c for c in _LETTER_RE.findall(target) if c in valid} @@ -256,7 +262,7 @@ def _make_generative_task(): evaluation_splits=["test"], few_shots_split=None, few_shots_select=None, - generation_size=16384, + generation_size=4096, metrics=[exo7_generative_f1_metric, exo7_generative_exact_metric], stop_sequence=[], version=0, From 17566ee0602a9ce7002349d622462ba99cd8f0e9 Mon Sep 17 00:00:00 2001 From: lduignan Date: Wed, 6 May 2026 16:19:02 +0200 Subject: [PATCH 06/11] reformat exo7 script --- community_tasks/exo7.py | 7 ++----- 1 file changed, 2 insertions(+), 5 deletions(-) diff --git a/community_tasks/exo7.py b/community_tasks/exo7.py index c96b134bf..e827663c6 100644 --- a/community_tasks/exo7.py +++ b/community_tasks/exo7.py @@ -164,8 +164,7 @@ def compute(self, model_response: ModelResponse, doc: Doc, **kwargs): # --- Prompt function --- INSTRUCTION = ( - "Pour la question suivante, une ou plusieurs propositions peuvent être correctes. " - "Évaluez chaque proposition." + "Pour la question suivante, une ou plusieurs propositions peuvent être correctes. Évaluez chaque proposition." ) @@ -207,9 +206,7 @@ def prompt_fn(line, task_name: str = None): correct_letters = [letters[i] for i, label in enumerate(labels) if label == 1] instruction = GENERATIVE_INSTRUCTION_TEMPLATE.format(valid_letters=", ".join(letters)) - choices_str = "\n".join( - f"{letter}) {choice.strip()}" for letter, choice in zip(letters, choices) - ) + choices_str = "\n".join(f"{letter}) {choice.strip()}" for letter, choice in zip(letters, choices)) query = f"{instruction}\n\n{line['question'].strip()}\n\n{choices_str}" doc = Doc( From 43682a58ca2cd13356f45dd5d9bec3f390e7667d Mon Sep 17 00:00:00 2001 From: lduignan Date: Wed, 6 May 2026 16:30:51 +0200 Subject: [PATCH 07/11] add copyright license to exo7.py --- community_tasks/exo7.py | 22 ++++++++++++++++++++++ 1 file changed, 22 insertions(+) diff --git a/community_tasks/exo7.py b/community_tasks/exo7.py index e827663c6..796204e1c 100644 --- a/community_tasks/exo7.py +++ b/community_tasks/exo7.py @@ -1,3 +1,25 @@ +# MIT License + +# Copyright (c) 2026 OpenLLM-France + +# Permission is hereby granted, free of charge, to any person obtaining a copy +# of this software and associated documentation files (the "Software"), to deal +# in the Software without restriction, including without limitation the rights +# to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +# copies of the Software, and to permit persons to whom the Software is +# furnished to do so, subject to the following conditions: + +# The above copyright notice and this permission notice shall be included in all +# copies or substantial portions of the Software. + +# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# SOFTWARE. + """ name: Exo7 From 5f19f35c97f2828bdfbcf318a3ec18dcee169d52 Mon Sep 17 00:00:00 2001 From: lduignan Date: Wed, 6 May 2026 17:38:45 +0200 Subject: [PATCH 08/11] Refactor Exo7MCMetric to improve logprob normalization using token and character lengths --- community_tasks/exo7.py | 27 ++++++++++++++++----------- 1 file changed, 16 insertions(+), 11 deletions(-) diff --git a/community_tasks/exo7.py b/community_tasks/exo7.py index 796204e1c..90e2fb8ab 100644 --- a/community_tasks/exo7.py +++ b/community_tasks/exo7.py @@ -49,6 +49,7 @@ import numpy as np from lighteval.metrics.metrics_sample import SampleLevelComputation +from lighteval.metrics.normalizations import LogProbCharNorm, LogProbTokenNorm, normalize_log_probs from lighteval.metrics.utils.metric_utils import SampleLevelMetric from lighteval.models.model_output import ModelResponse from lighteval.tasks.default_prompts import LETTER_INDICES @@ -73,19 +74,23 @@ class Exo7MCMetric(SampleLevelComputation): """ def compute(self, model_response: ModelResponse, doc: Doc, **kwargs): - logprobs = np.array(model_response.logprobs) - - # Length-normalize per choice. Prefer per-choice token counts from the - # model wrapper; fall back to character lengths of the formulation - # targets if the backend didn't populate output_tokens. - if model_response.output_tokens: - lengths = np.array([max(len(t), 1) for t in model_response.output_tokens]) + # Prefer per-choice token counts from the model wrapper; fall back to + # character normalization if the backend didn't populate output_tokens. + if model_response.output_tokens and all(len(t) > 0 for t in model_response.output_tokens): + normalization = LogProbTokenNorm() else: - lengths = np.array([max(len(c), 1) for c in doc.choices]) - - norm_logprobs = logprobs / lengths + normalization = LogProbCharNorm() + + norm_logprobs = np.array( + normalize_log_probs( + normalization, + choices_logprob=model_response.logprobs, + unconditioned_logprob=None, + choices_text=doc.choices, + choices_tokens=model_response.output_tokens, + ) + ) - # Stability shift + softmax probs = np.exp(norm_logprobs - np.max(norm_logprobs)) probs_norm = probs / np.sum(probs) From 4f7b7383c51268211d59e285c19f2a973ada00d1 Mon Sep 17 00:00:00 2001 From: lduignan Date: Wed, 6 May 2026 18:27:54 +0200 Subject: [PATCH 09/11] Refactor Exo7MCMetric to use both built in normalization methods --- community_tasks/exo7.py | 12 ++++-------- 1 file changed, 4 insertions(+), 8 deletions(-) diff --git a/community_tasks/exo7.py b/community_tasks/exo7.py index 90e2fb8ab..d9a310f93 100644 --- a/community_tasks/exo7.py +++ b/community_tasks/exo7.py @@ -73,17 +73,13 @@ class Exo7MCMetric(SampleLevelComputation): the total probability mass on the correct answers. """ - def compute(self, model_response: ModelResponse, doc: Doc, **kwargs): - # Prefer per-choice token counts from the model wrapper; fall back to - # character normalization if the backend didn't populate output_tokens. - if model_response.output_tokens and all(len(t) > 0 for t in model_response.output_tokens): - normalization = LogProbTokenNorm() - else: - normalization = LogProbCharNorm() + def __init__(self, normalization): + self.normalization = normalization + def compute(self, doc: Doc, model_response: ModelResponse, **kwargs): norm_logprobs = np.array( normalize_log_probs( - normalization, + self.normalization, choices_logprob=model_response.logprobs, unconditioned_logprob=None, choices_text=doc.choices, From 0c7414ce64ff9af3b019c8f363e9258ab3c9ac6f Mon Sep 17 00:00:00 2001 From: lduignan Date: Wed, 6 May 2026 18:46:28 +0200 Subject: [PATCH 10/11] rename metric --- community_tasks/exo7.py | 14 +++++++++++--- 1 file changed, 11 insertions(+), 3 deletions(-) diff --git a/community_tasks/exo7.py b/community_tasks/exo7.py index d9a310f93..06ab75d27 100644 --- a/community_tasks/exo7.py +++ b/community_tasks/exo7.py @@ -94,9 +94,17 @@ def compute(self, doc: Doc, model_response: ModelResponse, **kwargs): return float(np.sum(probs_norm[labels == 1])) -exo7_mc_metric = SampleLevelMetric( - metric_name="acc", - sample_level_fn=Exo7MCMetric(), +exo7_mc_metric_token = SampleLevelMetric( + metric_name="prob_mass_norm_token", + sample_level_fn=Exo7MCMetric(LogProbTokenNorm()), + category=SamplingMethod.LOGPROBS, + corpus_level_fn=np.mean, + higher_is_better=True, +) + +exo7_mc_metric_char = SampleLevelMetric( + metric_name="prob_mass_norm_char", + sample_level_fn=Exo7MCMetric(LogProbCharNorm()), category=SamplingMethod.LOGPROBS, corpus_level_fn=np.mean, higher_is_better=True, From b2d0f19ebaac8f0961e80c66362a09966727228f Mon Sep 17 00:00:00 2001 From: Liam <78303756+Lduignan1@users.noreply.github.com> Date: Tue, 9 Jun 2026 13:50:11 +0200 Subject: [PATCH 11/11] Update metrics in exo7.py to include token and char --- community_tasks/exo7.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/community_tasks/exo7.py b/community_tasks/exo7.py index 06ab75d27..b82534ae0 100644 --- a/community_tasks/exo7.py +++ b/community_tasks/exo7.py @@ -273,7 +273,7 @@ def _make_task(formulation): few_shots_split=None, few_shots_select=None, generation_size=1, - metrics=[exo7_mc_metric], + metrics=[exo7_mc_metric_token, exo7_mc_metric_char], stop_sequence=["\n"], version=0, )