Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
145 changes: 72 additions & 73 deletions evaluation_function/evaluation.py
Original file line number Diff line number Diff line change
Expand Up @@ -4,7 +4,7 @@
from lf_toolkit.evaluation import Result, Params
from lf_toolkit.parse.set import SetParser, LatexPrinter, SymPyBooleanTransformer, ASCIIPrinter, SymPyTransformer

from .parse import parse_with_feedback, FeedbackException
from .parse import parse_with_feedback

logger = logging.getLogger(__name__)
logging.basicConfig(level=logging.DEBUG, format="%(levelname)s [%(name)s] %(message)s")
Expand Down Expand Up @@ -35,6 +35,11 @@ def evaluation_function(
split into many) is entirely up to you. All that matters are the
return types and that evaluation_function() is the main function used
to output the evaluation response.

Raises `FeedbackException` (a `ValueError`) if the response cannot be
parsed, which lf_toolkit reports as an invalid submission (422), and
`RuntimeError` if the answer cannot be parsed, which is reported as an
internal error (500) since the fault lies with the task, not the student.
"""

logger.debug("evaluation_function called")
Expand All @@ -47,76 +52,70 @@ def evaluation_function(
# here we want to compare the response set with the example solution set.
# we have to do the following steps

is_latex = params.get("is_latex", False)
is_set_notation = params.get("is_set_notation", False)
transformer = SymPyTransformer() if is_set_notation else SymPyBooleanTransformer()
logger.debug("is_latex=%r", is_latex)
logger.debug("is_set_notation=%r", is_set_notation)

# 1. convert the `response`, which may be a latex string, to a sympy expression
logger.debug("parsing response...")
responseSet = parse_with_feedback(response, latex=is_latex)
logger.debug("responseSet=%r", responseSet)
responseSetSympy = transformer.transform(responseSet)
logger.debug("responseSetSympy=%r", responseSetSympy)

# 2. convert the `answer`, which may be a latex string, to a sympy expression
# TODO: what if answer is also in latex? how do we know?
logger.debug("parsing answer...")
try:
is_latex = params.get("is_latex", False)
is_set_notation = params.get("is_set_notation", False)
transformer = SymPyTransformer() if is_set_notation else SymPyBooleanTransformer()
logger.debug("is_latex=%r", is_latex)
logger.debug("is_set_notation=%r", is_set_notation)

# 1. convert the `response`, which may be a latex string, to a sympy expression
logger.debug("parsing response...")
responseSet = parse_with_feedback(response, latex=is_latex)
logger.debug("responseSet=%r", responseSet)
responseSetSympy = transformer.transform(responseSet)
logger.debug("responseSetSympy=%r", responseSetSympy)

# 2. convert the `answer`, which may be a latex string, to a sympy expression
# TODO: what if answer is also in latex? how do we know?
logger.debug("parsing answer...")
try:
answerSet = parser.parse(answer, latex=False)
except Exception as e:
logger.error("failed to parse answer: type=%s value=%r error=%r", type(answer).__name__, answer, e)
raise FeedbackException() from e
logger.debug("answerSet=%r", answerSet)
answerSetSympy = transformer.transform(answerSet)
logger.debug("answerSetSympy=%r", answerSetSympy)

# 3. compare the two sympy expressions w/ simplification enabled.
# If they are equal, the sets produced by the two expressions are
# semantically equal. However, the expressions may not be equal.
if is_set_notation:
semantic_equal = responseSetSympy == answerSetSympy
else:
semantic_equal = simplify_logic(Equivalent(responseSetSympy, answerSetSympy)) == True
logger.debug("semantic_equal=%r", semantic_equal)

# 4. compare the two sympy expressions w/ simplifaction disabled.
# If they are equal, the expressions are also equal in syntax.
# This respects laws of commutativity, e.g. A u B == B u A.
syntactic_equal = responseSetSympy == answerSetSympy
logger.debug("syntactic_equal=%r", syntactic_equal)

enforce_expression_equality = params.get("enforce_expression_equality", False)
logger.debug("enforce_expression_equality=%r", enforce_expression_equality)

# 5. `is_correct` is True, iff 3) is True, and either 4) or `enforce_expression_equality` is True
is_correct = semantic_equal and (syntactic_equal or not enforce_expression_equality)
logger.debug("is_correct=%r", is_correct)

feedback_items=[]

if semantic_equal and not syntactic_equal and enforce_expression_equality:
feedback_items.append(("syntactic_equality", "The expressions are not equal syntacitcally."))
elif not semantic_equal:
feedback_items.append(("semantic_equality", "The expressions are not equal."))

latexPrinter = LatexPrinter()
latex = latexPrinter.print(responseSet)

asciiPrinter = ASCIIPrinter()
ascii = asciiPrinter.print(responseSet)

return Result(
is_correct=is_correct,
latex=latex,
simplified=ascii,
feedback_items=feedback_items,
)
except FeedbackException as e:
logger.error("FeedbackException: %r", e)
return Result(
is_correct=False,
feedback_items=[("parse_error", str(e))]
)
answerSet = parser.parse(answer, latex=False)
except Exception as e:
logger.error("failed to parse answer: type=%s value=%r error=%r", type(answer).__name__, answer, e)
# not a ValueError, so this is reported as an internal error rather than an invalid submission
raise RuntimeError(f"Failed to parse answer: {e}") from e
logger.debug("answerSet=%r", answerSet)
answerSetSympy = transformer.transform(answerSet)
logger.debug("answerSetSympy=%r", answerSetSympy)

# 3. compare the two sympy expressions w/ simplification enabled.
# If they are equal, the sets produced by the two expressions are
# semantically equal. However, the expressions may not be equal.
if is_set_notation:
semantic_equal = responseSetSympy == answerSetSympy
else:
semantic_equal = simplify_logic(Equivalent(responseSetSympy, answerSetSympy)) == True
logger.debug("semantic_equal=%r", semantic_equal)

# 4. compare the two sympy expressions w/ simplifaction disabled.
# If they are equal, the expressions are also equal in syntax.
# This respects laws of commutativity, e.g. A u B == B u A.
syntactic_equal = responseSetSympy == answerSetSympy
logger.debug("syntactic_equal=%r", syntactic_equal)

enforce_expression_equality = params.get("enforce_expression_equality", False)
logger.debug("enforce_expression_equality=%r", enforce_expression_equality)

# 5. `is_correct` is True, iff 3) is True, and either 4) or `enforce_expression_equality` is True
is_correct = semantic_equal and (syntactic_equal or not enforce_expression_equality)
logger.debug("is_correct=%r", is_correct)

feedback_items=[]

if semantic_equal and not syntactic_equal and enforce_expression_equality:
feedback_items.append(("syntactic_equality", "The expressions are not equal syntacitcally."))
elif not semantic_equal:
feedback_items.append(("semantic_equality", "The expressions are not equal."))

latexPrinter = LatexPrinter()
latex = latexPrinter.print(responseSet)

asciiPrinter = ASCIIPrinter()
ascii = asciiPrinter.print(responseSet)

return Result(
is_correct=is_correct,
latex=latex,
simplified=ascii,
feedback_items=feedback_items,
)
21 changes: 13 additions & 8 deletions evaluation_function/evaluation_test.py
Original file line number Diff line number Diff line change
Expand Up @@ -100,14 +100,19 @@ def test_returns_is_correct_false(self):
self.assertEqual(result.get("response_latex"), "A \\cap B")
self.assertTrue(result.get("feedback"))

def test_returns_is_correct_false_not_parseable(self):
response, answer, params = "", "A u B", Params()

result = evaluation_function(response, answer, params).to_dict()

self.assertEqual(result.get("is_correct"), False)
self.assertEqual(result.get("response_latex"), None)
self.assertTrue(result.get("feedback"))
def test_unparseable_response_raises_value_error(self):
# a ValueError is reported by lf_toolkit as an invalid submission (422)
for response in ["", "A n "]:
with self.subTest(response=response):
with self.assertRaises(ValueError):
evaluation_function(response, "A u B", Params())

def test_unparseable_answer_raises_non_value_error(self):
# must not be a ValueError, so it is reported as an internal error (500)
with self.assertRaises(RuntimeError) as ctx:
evaluation_function("A u B", "A n ", Params())

self.assertNotIsInstance(ctx.exception, ValueError)

def test_syntactic_returns_is_correct_true_commutativity(self):
response, answer, params = "A u B", "B u A", Params(enforce_expression_equality=True)
Expand Down
4 changes: 3 additions & 1 deletion evaluation_function/parse.py
Original file line number Diff line number Diff line change
@@ -1,6 +1,8 @@
from lf_toolkit.parse.set import SetParser, ParseError

class FeedbackException(Exception):
# Subclasses ValueError so that, when raised from the evaluation function,
# lf_toolkit reports it as an invalid submission (422) rather than a 500.
class FeedbackException(ValueError):

def __str__(self):
if isinstance(self.__cause__, ParseError):
Expand Down
17 changes: 9 additions & 8 deletions poetry.lock

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

2 changes: 1 addition & 1 deletion pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -12,7 +12,7 @@ evaluation_function = "evaluation_function.main:main"
[tool.poetry.dependencies]
python = "^3.11"
typing_extensions = "^4.12.2"
lf_toolkit = { git = "https://github.com/lambda-feedback/toolkit-python.git", tag = "v1.0.3", extras = [
lf_toolkit = { git = "https://github.com/lambda-feedback/toolkit-python.git", tag = "v1.2.0", extras = [
"parsing",
"ipc",
] }
Expand Down
Loading