From 3905f13a9437d6dff2d01bf78b9add2866da56bc Mon Sep 17 00:00:00 2001 From: Marcus Messer Date: Mon, 5 Oct 2026 18:40:49 +0100 Subject: [PATCH 1/4] Raise on unparseable submissions so they surface as 422 Unparseable responses previously returned a 200 with is_correct=False and parse_error feedback, so lf_toolkit's invalid-submission handling never fired. FeedbackException now subclasses ValueError and is no longer caught, so lf_toolkit reports it as an invalid submission (422). An unparseable answer is a fault in the task rather than the student's submission, so it raises RuntimeError instead and is reported as a 500. Co-Authored-By: Claude Opus 5.5 --- evaluation_function/evaluation.py | 145 ++++++++++++------------- evaluation_function/evaluation_test.py | 21 ++-- evaluation_function/parse.py | 4 +- 3 files changed, 88 insertions(+), 82 deletions(-) diff --git a/evaluation_function/evaluation.py b/evaluation_function/evaluation.py index d436aec..6ad44f7 100755 --- a/evaluation_function/evaluation.py +++ b/evaluation_function/evaluation.py @@ -4,7 +4,7 @@ from lf_toolkit.evaluation import Result, Params from lf_toolkit.parse.set import SetParser, LatexPrinter, SymPyBooleanTransformer, ASCIIPrinter, SymPyTransformer -from .parse import parse_with_feedback, FeedbackException +from .parse import parse_with_feedback logger = logging.getLogger(__name__) logging.basicConfig(level=logging.DEBUG, format="%(levelname)s [%(name)s] %(message)s") @@ -35,6 +35,11 @@ def evaluation_function( split into many) is entirely up to you. All that matters are the return types and that evaluation_function() is the main function used to output the evaluation response. + + Raises `FeedbackException` (a `ValueError`) if the response cannot be + parsed, which lf_toolkit reports as an invalid submission (422), and + `RuntimeError` if the answer cannot be parsed, which is reported as an + internal error (500) since the fault lies with the task, not the student. """ logger.debug("evaluation_function called") @@ -47,76 +52,70 @@ def evaluation_function( # here we want to compare the response set with the example solution set. # we have to do the following steps + is_latex = params.get("is_latex", False) + is_set_notation = params.get("is_set_notation", False) + transformer = SymPyTransformer() if is_set_notation else SymPyBooleanTransformer() + logger.debug("is_latex=%r", is_latex) + logger.debug("is_set_notation=%r", is_set_notation) + + # 1. convert the `response`, which may be a latex string, to a sympy expression + logger.debug("parsing response...") + responseSet = parse_with_feedback(response, latex=is_latex) + logger.debug("responseSet=%r", responseSet) + responseSetSympy = transformer.transform(responseSet) + logger.debug("responseSetSympy=%r", responseSetSympy) + + # 2. convert the `answer`, which may be a latex string, to a sympy expression + # TODO: what if answer is also in latex? how do we know? + logger.debug("parsing answer...") try: - is_latex = params.get("is_latex", False) - is_set_notation = params.get("is_set_notation", False) - transformer = SymPyTransformer() if is_set_notation else SymPyBooleanTransformer() - logger.debug("is_latex=%r", is_latex) - logger.debug("is_set_notation=%r", is_set_notation) - - # 1. convert the `response`, which may be a latex string, to a sympy expression - logger.debug("parsing response...") - responseSet = parse_with_feedback(response, latex=is_latex) - logger.debug("responseSet=%r", responseSet) - responseSetSympy = transformer.transform(responseSet) - logger.debug("responseSetSympy=%r", responseSetSympy) - - # 2. convert the `answer`, which may be a latex string, to a sympy expression - # TODO: what if answer is also in latex? how do we know? - logger.debug("parsing answer...") - try: - answerSet = parser.parse(answer, latex=False) - except Exception as e: - logger.error("failed to parse answer: type=%s value=%r error=%r", type(answer).__name__, answer, e) - raise FeedbackException() from e - logger.debug("answerSet=%r", answerSet) - answerSetSympy = transformer.transform(answerSet) - logger.debug("answerSetSympy=%r", answerSetSympy) - - # 3. compare the two sympy expressions w/ simplification enabled. - # If they are equal, the sets produced by the two expressions are - # semantically equal. However, the expressions may not be equal. - if is_set_notation: - semantic_equal = responseSetSympy == answerSetSympy - else: - semantic_equal = simplify_logic(Equivalent(responseSetSympy, answerSetSympy)) == True - logger.debug("semantic_equal=%r", semantic_equal) - - # 4. compare the two sympy expressions w/ simplifaction disabled. - # If they are equal, the expressions are also equal in syntax. - # This respects laws of commutativity, e.g. A u B == B u A. - syntactic_equal = responseSetSympy == answerSetSympy - logger.debug("syntactic_equal=%r", syntactic_equal) - - enforce_expression_equality = params.get("enforce_expression_equality", False) - logger.debug("enforce_expression_equality=%r", enforce_expression_equality) - - # 5. `is_correct` is True, iff 3) is True, and either 4) or `enforce_expression_equality` is True - is_correct = semantic_equal and (syntactic_equal or not enforce_expression_equality) - logger.debug("is_correct=%r", is_correct) - - feedback_items=[] - - if semantic_equal and not syntactic_equal and enforce_expression_equality: - feedback_items.append(("syntactic_equality", "The expressions are not equal syntacitcally.")) - elif not semantic_equal: - feedback_items.append(("semantic_equality", "The expressions are not equal.")) - - latexPrinter = LatexPrinter() - latex = latexPrinter.print(responseSet) - - asciiPrinter = ASCIIPrinter() - ascii = asciiPrinter.print(responseSet) - - return Result( - is_correct=is_correct, - latex=latex, - simplified=ascii, - feedback_items=feedback_items, - ) - except FeedbackException as e: - logger.error("FeedbackException: %r", e) - return Result( - is_correct=False, - feedback_items=[("parse_error", str(e))] - ) + answerSet = parser.parse(answer, latex=False) + except Exception as e: + logger.error("failed to parse answer: type=%s value=%r error=%r", type(answer).__name__, answer, e) + # not a ValueError, so this is reported as an internal error rather than an invalid submission + raise RuntimeError(f"Failed to parse answer: {e}") from e + logger.debug("answerSet=%r", answerSet) + answerSetSympy = transformer.transform(answerSet) + logger.debug("answerSetSympy=%r", answerSetSympy) + + # 3. compare the two sympy expressions w/ simplification enabled. + # If they are equal, the sets produced by the two expressions are + # semantically equal. However, the expressions may not be equal. + if is_set_notation: + semantic_equal = responseSetSympy == answerSetSympy + else: + semantic_equal = simplify_logic(Equivalent(responseSetSympy, answerSetSympy)) == True + logger.debug("semantic_equal=%r", semantic_equal) + + # 4. compare the two sympy expressions w/ simplifaction disabled. + # If they are equal, the expressions are also equal in syntax. + # This respects laws of commutativity, e.g. A u B == B u A. + syntactic_equal = responseSetSympy == answerSetSympy + logger.debug("syntactic_equal=%r", syntactic_equal) + + enforce_expression_equality = params.get("enforce_expression_equality", False) + logger.debug("enforce_expression_equality=%r", enforce_expression_equality) + + # 5. `is_correct` is True, iff 3) is True, and either 4) or `enforce_expression_equality` is True + is_correct = semantic_equal and (syntactic_equal or not enforce_expression_equality) + logger.debug("is_correct=%r", is_correct) + + feedback_items=[] + + if semantic_equal and not syntactic_equal and enforce_expression_equality: + feedback_items.append(("syntactic_equality", "The expressions are not equal syntacitcally.")) + elif not semantic_equal: + feedback_items.append(("semantic_equality", "The expressions are not equal.")) + + latexPrinter = LatexPrinter() + latex = latexPrinter.print(responseSet) + + asciiPrinter = ASCIIPrinter() + ascii = asciiPrinter.print(responseSet) + + return Result( + is_correct=is_correct, + latex=latex, + simplified=ascii, + feedback_items=feedback_items, + ) diff --git a/evaluation_function/evaluation_test.py b/evaluation_function/evaluation_test.py index 2783c42..8b82de7 100755 --- a/evaluation_function/evaluation_test.py +++ b/evaluation_function/evaluation_test.py @@ -100,14 +100,19 @@ def test_returns_is_correct_false(self): self.assertEqual(result.get("response_latex"), "A \\cap B") self.assertTrue(result.get("feedback")) - def test_returns_is_correct_false_not_parseable(self): - response, answer, params = "", "A u B", Params() - - result = evaluation_function(response, answer, params).to_dict() - - self.assertEqual(result.get("is_correct"), False) - self.assertEqual(result.get("response_latex"), None) - self.assertTrue(result.get("feedback")) + def test_unparseable_response_raises_value_error(self): + # a ValueError is reported by lf_toolkit as an invalid submission (422) + for response in ["", "A n "]: + with self.subTest(response=response): + with self.assertRaises(ValueError): + evaluation_function(response, "A u B", Params()) + + def test_unparseable_answer_raises_non_value_error(self): + # must not be a ValueError, so it is reported as an internal error (500) + with self.assertRaises(RuntimeError) as ctx: + evaluation_function("A u B", "A n ", Params()) + + self.assertNotIsInstance(ctx.exception, ValueError) def test_syntactic_returns_is_correct_true_commutativity(self): response, answer, params = "A u B", "B u A", Params(enforce_expression_equality=True) diff --git a/evaluation_function/parse.py b/evaluation_function/parse.py index 9bc8672..6ac70e9 100644 --- a/evaluation_function/parse.py +++ b/evaluation_function/parse.py @@ -1,6 +1,8 @@ from lf_toolkit.parse.set import SetParser, ParseError -class FeedbackException(Exception): +# Subclasses ValueError so that, when raised from the evaluation function, +# lf_toolkit reports it as an invalid submission (422) rather than a 500. +class FeedbackException(ValueError): def __str__(self): if isinstance(self.__cause__, ParseError): From fe0adeba368ed5b97ec1a88124eda6ff675a0eef Mon Sep 17 00:00:00 2001 From: Marcus Messer Date: Mon, 5 Oct 2026 18:54:51 +0100 Subject: [PATCH 2/4] Point lf_toolkit at the invalid-submission-error commit Pins lf_toolkit to toolkit-python 9d89276 (feature/invalid-submission-error), which maps a ValueError raised by the evaluation function to a JSON-RPC 422 error, so unparseable submissions can be tested end to end on staging. Co-Authored-By: Claude Opus 5.5 --- poetry.lock | 17 +++++++++-------- pyproject.toml | 2 +- 2 files changed, 10 insertions(+), 9 deletions(-) diff --git a/poetry.lock b/poetry.lock index 5eb78af..3474cba 100644 --- a/poetry.lock +++ b/poetry.lock @@ -1,4 +1,4 @@ -# This file is automatically @generated by Poetry 2.1.3 and should not be changed by hand. +# This file is automatically @generated by Poetry 2.4.3 and should not be changed by hand. [[package]] name = "annotated-types" @@ -920,7 +920,7 @@ files = [ [package.dependencies] attrs = ">=22.2.0" -jsonschema-specifications = ">=2023.03.6" +jsonschema-specifications = ">=2023.3.6" referencing = ">=0.28.4" rpds-py = ">=0.7.1" @@ -1038,6 +1038,7 @@ sympy = ">=1.12,<2.0" ujson = "5.10.0" [package.extras] +gcs = ["google-cloud-storage (>=2.18,<3.0)"] http = ["fastapi (>=0.115.0,<0.116.0)"] ipc = ["pywin32 (>=306,<307) ; sys_platform == \"win32\""] parsing = ["antlr4-python3-runtime (==4.13.2)", "lark (==1.2.2)", "latex2sympy @ git+https://github.com/purdue-tlt/latex2sympy.git@1.12.0"] @@ -1045,8 +1046,8 @@ parsing = ["antlr4-python3-runtime (==4.13.2)", "lark (==1.2.2)", "latex2sympy @ [package.source] type = "git" url = "https://github.com/lambda-feedback/toolkit-python.git" -reference = "v1.0.3" -resolved_reference = "8a687d35284c156f045dde4da15a2e717112c9de" +reference = "9d89276f4cf3dd943003aa765cb917e5ca20cfe0" +resolved_reference = "9d89276f4cf3dd943003aa765cb917e5ca20cfe0" [[package]] name = "mccabe" @@ -1989,10 +1990,10 @@ files = [ ] [package.dependencies] -botocore = ">=1.37.4,<2.0a.0" +botocore = ">=1.37.4,<2.0a0" [package.extras] -crt = ["botocore[crt] (>=1.37.4,<2.0a.0)"] +crt = ["botocore[crt] (>=1.37.4,<2.0a0)"] [[package]] name = "secretstorage" @@ -2441,9 +2442,9 @@ files = [ ] [package.extras] -cffi = ["cffi (>=1.17,<2.0) ; platform_python_implementation != \"PyPy\" and python_version < \"3.14\"", "cffi (>=2.0.0b) ; platform_python_implementation != \"PyPy\" and python_version >= \"3.14\""] +cffi = ["cffi (>=1.17,<2.0) ; platform_python_implementation != \"PyPy\" and python_version < \"3.14\"", "cffi (>=2.0.0b0) ; platform_python_implementation != \"PyPy\" and python_version >= \"3.14\""] [metadata] lock-version = "2.1" python-versions = "^3.11" -content-hash = "1075e2f154de627eeb1644b1145ff179453ff7c4dccb98a0bc0de1a46ffd84d1" +content-hash = "efae3e5cf2d15fe0cf2029ef63d96888d46c6bacd6ceca2cf28ae03e0ac56369" diff --git a/pyproject.toml b/pyproject.toml index 71dffc0..d87e3a2 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -12,7 +12,7 @@ evaluation_function = "evaluation_function.main:main" [tool.poetry.dependencies] python = "^3.11" typing_extensions = "^4.12.2" -lf_toolkit = { git = "https://github.com/lambda-feedback/toolkit-python.git", tag = "v1.0.3", extras = [ +lf_toolkit = { git = "https://github.com/lambda-feedback/toolkit-python.git", rev = "9d89276f4cf3dd943003aa765cb917e5ca20cfe0", extras = [ "parsing", "ipc", ] } From d8f6a29a06d35578eec3b8f328294d5727f76fbc Mon Sep 17 00:00:00 2001 From: Marcus Messer Date: Mon, 5 Oct 2026 18:55:52 +0100 Subject: [PATCH 3/4] Update Docker base image to `python:shimmy-test-3.12` Co-Authored-By: Claude Opus 5.5 --- Dockerfile | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/Dockerfile b/Dockerfile index 1f73b32..26fdfb5 100755 --- a/Dockerfile +++ b/Dockerfile @@ -1,4 +1,4 @@ -ARG BASE_VERSION=python:edge-3.12 +ARG BASE_VERSION=python:shimmy-test-3.12 FROM ghcr.io/lambda-feedback/evaluation-function-base/$BASE_VERSION AS builder RUN pip install poetry==1.8.3 From c5eebf66d8977f04a5cb35ba9091e735262db6af Mon Sep 17 00:00:00 2001 From: Marcus Messer Date: Mon, 5 Oct 2026 19:14:10 +0100 Subject: [PATCH 4/4] Use lf_toolkit v1.2.0 and the edge base image toolkit-python v1.2.0 and shimmy main now include the invalid-submission (422) handling, so switch from the test builds to the released toolkit tag and the edge base image. Co-Authored-By: Claude Opus 5.5 --- Dockerfile | 2 +- poetry.lock | 6 +++--- pyproject.toml | 2 +- 3 files changed, 5 insertions(+), 5 deletions(-) diff --git a/Dockerfile b/Dockerfile index 26fdfb5..1f73b32 100755 --- a/Dockerfile +++ b/Dockerfile @@ -1,4 +1,4 @@ -ARG BASE_VERSION=python:shimmy-test-3.12 +ARG BASE_VERSION=python:edge-3.12 FROM ghcr.io/lambda-feedback/evaluation-function-base/$BASE_VERSION AS builder RUN pip install poetry==1.8.3 diff --git a/poetry.lock b/poetry.lock index 3474cba..595fcea 100644 --- a/poetry.lock +++ b/poetry.lock @@ -1046,8 +1046,8 @@ parsing = ["antlr4-python3-runtime (==4.13.2)", "lark (==1.2.2)", "latex2sympy @ [package.source] type = "git" url = "https://github.com/lambda-feedback/toolkit-python.git" -reference = "9d89276f4cf3dd943003aa765cb917e5ca20cfe0" -resolved_reference = "9d89276f4cf3dd943003aa765cb917e5ca20cfe0" +reference = "v1.2.0" +resolved_reference = "df5efa3735df67a75992e95b3a1c2e6df16a6752" [[package]] name = "mccabe" @@ -2447,4 +2447,4 @@ cffi = ["cffi (>=1.17,<2.0) ; platform_python_implementation != \"PyPy\" and pyt [metadata] lock-version = "2.1" python-versions = "^3.11" -content-hash = "efae3e5cf2d15fe0cf2029ef63d96888d46c6bacd6ceca2cf28ae03e0ac56369" +content-hash = "9ba667bb16098f2642edb6c8a2c19f50e419847beee907cf2d962a49826522d4" diff --git a/pyproject.toml b/pyproject.toml index d87e3a2..e16cb28 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -12,7 +12,7 @@ evaluation_function = "evaluation_function.main:main" [tool.poetry.dependencies] python = "^3.11" typing_extensions = "^4.12.2" -lf_toolkit = { git = "https://github.com/lambda-feedback/toolkit-python.git", rev = "9d89276f4cf3dd943003aa765cb917e5ca20cfe0", extras = [ +lf_toolkit = { git = "https://github.com/lambda-feedback/toolkit-python.git", tag = "v1.2.0", extras = [ "parsing", "ipc", ] }