From 946a5228c544ffa121fd312596fb83a1aea11595 Mon Sep 17 00:00:00 2001 From: Eugenio Zuccarelli <11176606+jayzuccarelli@users.noreply.github.com> Date: Sat, 11 Jul 2026 13:06:00 +0000 Subject: [PATCH 1/3] Use ast.literal_eval instead of eval in hellaswag_arabic_pfn The endings field comes from a community dataset on the Hub; eval() executes arbitrary Python from it. ast.literal_eval parses the same string-encoded lists safely and raises on anything else, matching how other tasks already parse such fields. Fixes #1293 --- .../tasks/multilingual/tasks/arabic.py | 3 +- tests/unit/tasks/test_arabic_tasks.py | 51 +++++++++++++++++++ 2 files changed, 53 insertions(+), 1 deletion(-) create mode 100644 tests/unit/tasks/test_arabic_tasks.py diff --git a/src/lighteval/tasks/multilingual/tasks/arabic.py b/src/lighteval/tasks/multilingual/tasks/arabic.py index 2d78f9887..226e8bac5 100644 --- a/src/lighteval/tasks/multilingual/tasks/arabic.py +++ b/src/lighteval/tasks/multilingual/tasks/arabic.py @@ -17,6 +17,7 @@ paper: """ +import ast import random import re from string import ascii_uppercase @@ -600,7 +601,7 @@ def copa_arabic_pfn(line, task_name: str = None): def hellaswag_arabic_pfn(line, task_name: str = None): ctx = re.sub(r"\[.*?\]", "", line["ctx"]) # Remove latin words within brackets endings = [ - re.sub(r"\[.*?\]", "", e) for e in eval(line["endings"]) + re.sub(r"\[.*?\]", "", e) for e in ast.literal_eval(line["endings"]) ] # endings is a string representation of a list answer_index = line["label"] instruction = "بناء على السياق التالي، اختر النهاية الصحيحة من الاقتراحات التالية" diff --git a/tests/unit/tasks/test_arabic_tasks.py b/tests/unit/tasks/test_arabic_tasks.py new file mode 100644 index 000000000..25ad9eefc --- /dev/null +++ b/tests/unit/tasks/test_arabic_tasks.py @@ -0,0 +1,51 @@ +# MIT License + +# Copyright (c) 2024 The HuggingFace Team + +# Permission is hereby granted, free of charge, to any person obtaining a copy +# of this software and associated documentation files (the "Software"), to deal +# in the Software without restriction, including without limitation the rights +# to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +# copies of the Software, and to permit persons to whom the Software is +# furnished to do so, subject to the following conditions: + +# The above copyright notice and this permission notice shall be included in all +# copies or substantial portions of the Software. + +# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# SOFTWARE. + +import pytest + +from lighteval.tasks.multilingual.tasks.arabic import hellaswag_arabic_pfn + + +def test_hellaswag_arabic_pfn_parses_endings(): + line = { + "ctx": "sentence one [latin] and more", + "endings": "['ending one [x]', 'ending two']", + "label": 1, + } + + doc = hellaswag_arabic_pfn(line, "test_task") + + assert doc.choices == ["ending one ", "ending two"] + assert doc.gold_index == 1 + + +def test_hellaswag_arabic_pfn_rejects_non_literal_endings(): + # endings comes from a hub dataset; anything that is not a plain literal + # must raise instead of being evaluated + line = { + "ctx": "sentence", + "endings": "__import__('os').system('echo pwned')", + "label": 0, + } + + with pytest.raises(ValueError): + hellaswag_arabic_pfn(line, "test_task") From 0461865c15a84b9379d9b65a5edb3ddd44957529 Mon Sep 17 00:00:00 2001 From: "Eugenio \"Jay\" Zuccarelli" <11176606+jayzuccarelli@users.noreply.github.com> Date: Tue, 28 Jul 2026 22:07:40 +0000 Subject: [PATCH 2/3] Widen rejection assertion to SyntaxError and pin no-execution with a side-effect test --- tests/unit/tasks/test_arabic_tasks.py | 18 ++++++++++++++++-- 1 file changed, 16 insertions(+), 2 deletions(-) diff --git a/tests/unit/tasks/test_arabic_tasks.py b/tests/unit/tasks/test_arabic_tasks.py index 25ad9eefc..31b391dda 100644 --- a/tests/unit/tasks/test_arabic_tasks.py +++ b/tests/unit/tasks/test_arabic_tasks.py @@ -40,12 +40,26 @@ def test_hellaswag_arabic_pfn_parses_endings(): def test_hellaswag_arabic_pfn_rejects_non_literal_endings(): # endings comes from a hub dataset; anything that is not a plain literal - # must raise instead of being evaluated + # must raise instead of being evaluated. literal_eval raises ValueError on + # non-literal nodes but SyntaxError on malformed input (e.g. a truncated + # field), so both count as rejection. line = { "ctx": "sentence", "endings": "__import__('os').system('echo pwned')", "label": 0, } - with pytest.raises(ValueError): + with pytest.raises((ValueError, SyntaxError)): hellaswag_arabic_pfn(line, "test_task") + + +def test_hellaswag_arabic_pfn_does_not_execute_endings(): + # pytest.raises alone would also pass if something ran before raising; + # the side effect pins down that nothing in the field executes at all + executed = [] + line = {"ctx": "s", "endings": "[executed.append(1)]", "label": 0} + + with pytest.raises((ValueError, SyntaxError)): + hellaswag_arabic_pfn(line, "test_task") + + assert executed == [] From 3ba7b0144f8583f95a22d86fe61a731f75cbe3d7 Mon Sep 17 00:00:00 2001 From: "Eugenio \"Jay\" Zuccarelli" <11176606+jayzuccarelli@users.noreply.github.com> Date: Tue, 28 Jul 2026 23:14:11 +0000 Subject: [PATCH 3/3] Make the non-execution test actually falsifiable via a builtins-reachable side effect --- tests/unit/tasks/test_arabic_tasks.py | 24 +++++++++++++++--------- 1 file changed, 15 insertions(+), 9 deletions(-) diff --git a/tests/unit/tasks/test_arabic_tasks.py b/tests/unit/tasks/test_arabic_tasks.py index 31b391dda..9b2210ec8 100644 --- a/tests/unit/tasks/test_arabic_tasks.py +++ b/tests/unit/tasks/test_arabic_tasks.py @@ -20,6 +20,8 @@ # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE # SOFTWARE. +import sys + import pytest from lighteval.tasks.multilingual.tasks.arabic import hellaswag_arabic_pfn @@ -54,12 +56,16 @@ def test_hellaswag_arabic_pfn_rejects_non_literal_endings(): def test_hellaswag_arabic_pfn_does_not_execute_endings(): - # pytest.raises alone would also pass if something ran before raising; - # the side effect pins down that nothing in the field executes at all - executed = [] - line = {"ctx": "s", "endings": "[executed.append(1)]", "label": 0} - - with pytest.raises((ValueError, SyntaxError)): - hellaswag_arabic_pfn(line, "test_task") - - assert executed == [] + # pytest.raises alone would also pass if something ran before raising. The payload has to + # reach its side effect through builtins, because eval() here runs against arabic.py's module + # globals and cannot see anything defined in this test. + marker = "lighteval_pwned_by_dataset_field" + sys.modules.pop(marker, None) + line = {"ctx": "s", "endings": f"[__import__('sys').modules.setdefault({marker!r}, 1)]", "label": 0} + + try: + with pytest.raises((ValueError, SyntaxError)): + hellaswag_arabic_pfn(line, "test_task") + assert marker not in sys.modules + finally: + sys.modules.pop(marker, None)