diff --git a/.github/workflows/test.yml b/.github/workflows/test.yml index 0d93cef..42b4048 100644 --- a/.github/workflows/test.yml +++ b/.github/workflows/test.yml @@ -22,6 +22,15 @@ jobs: poetry install --with dev --all-extras - name: Install Pandoc # apt version seems too old uses: r-lib/actions/setup-pandoc@v2 + # The validator compiles a set the way lambda-feedback/PDF-generator does, so it + # needs what that image installs: xelatex with braket, cancel, xeCJK and the + # Noto Sans fonts the template sets as its main and CJK fonts. + - name: Install XeLaTeX + run: | + sudo apt-get update + sudo apt-get install -y --no-install-recommends \ + texlive-xetex texlive-latex-recommended texlive-latex-extra \ + texlive-science texlive-lang-chinese fonts-noto-core fonts-noto-cjk - name: Linting Checks run: | poetry run black --check . diff --git a/docs/source/quickstart.md b/docs/source/quickstart.md index acb0eb1..03fbe78 100644 --- a/docs/source/quickstart.md +++ b/docs/source/quickstart.md @@ -75,6 +75,8 @@ By default, this generates an `out` directory in the same place that the command Before writing anything, in2lambda prints the problems it can detect that would stop the set importing or make it render wrongly — an answer that doesn't fit the box marking it, a figure the export won't contain, maths that KaTeX can't display. Each names the question, part and field to go and look at. They are warnings rather than errors: the `out` directory is written either way, since a problem found here may well be deliberate. +With [xelatex](https://tug.org/texlive/) installed alongside pandoc, the set is also compiled the way Lambda Feedback makes a PDF of it, and any LaTeX error names the field it is in. Without it, one warning says which packages to install instead. + Check the [command line tool reference](reference/command-line) for more information. ## 3. Import into Lambda Feedback diff --git a/in2lambda/api/set.py b/in2lambda/api/set.py index 7dc904e..09b0e68 100644 --- a/in2lambda/api/set.py +++ b/in2lambda/api/set.py @@ -103,12 +103,16 @@ def increment_current_question(self) -> None: """ self._current_question_index += 1 - def problems(self) -> list[Problem]: + def problems(self, compile: bool = True) -> list[Problem]: r"""Everything in2lambda can tell Lambda Feedback would refuse or render wrongly. This is a report, not a refusal: the set can still be written out, since a problem found here may well be deliberate. + Args: + compile: Whether to also compile the set as Lambda Feedback's PDF generator + will, which needs pandoc and xelatex installed. + Returns: One :class:`~in2lambda.api.problem.Problem` per problem found, each naming the question, part and field to go and look at. @@ -118,14 +122,14 @@ def problems(self) -> list[Problem]: >>> s = Set() >>> s.add_question("Momentum", "The rocket is at $45^\\circ$.") >>> s.current_question.images.append("no_such_file.png") - >>> for problem in s.problems(): + >>> for problem in s.problems(compile=False): ... print(problem) Question 1 "Momentum", main text: ^\circ does not display; write the degree sign ° instead Question 1 "Momentum": there is no image file at no_such_file.png """ from in2lambda.validation import validate - return validate(self) + return validate(self, compile) def to_json(self, output_dir: str) -> None: """Turns this set into Lambda Feedback JSON/ZIP files. diff --git a/in2lambda/validation/__init__.py b/in2lambda/validation/__init__.py index 18ef5e4..1d42e67 100644 --- a/in2lambda/validation/__init__.py +++ b/in2lambda/validation/__init__.py @@ -18,6 +18,7 @@ from in2lambda.api.response_area import ResponseArea from in2lambda.api.set import Set from in2lambda.katex_convert.katex_convert import unsupported_commands +from in2lambda.validation import pdf from in2lambda.validation.delimiters import MathDelimiterError, math_delimiter_checker __all__ = ["MathDelimiterError", "Problem", "math_delimiter_checker", "validate"] @@ -34,11 +35,14 @@ """``^\\circ``, with or without braces around it.""" -def validate(question_set: Set) -> list[Problem]: +def validate(question_set: Set, compile: bool = True) -> list[Problem]: r"""Everything in2lambda can tell is wrong with a set, in the order it is written. Args: question_set: The set about to be exported. + compile: Whether to also compile the set as Lambda Feedback's PDF generator + will, which needs pandoc and xelatex - see + :mod:`in2lambda.validation.pdf`. Returns: One :class:`~in2lambda.api.problem.Problem` per problem found, each naming the @@ -50,17 +54,27 @@ def validate(question_set: Set) -> list[Problem]: >>> from in2lambda.validation import validate >>> s = Set() >>> s.add_question("Angles", "Turn through $90^\\circ$.") - >>> [str(problem) for problem in validate(s)] + >>> [str(problem) for problem in validate(s, compile=False)] ['Question 1 "Angles", main text: ^\\circ does not display; write the degree sign ° instead'] """ problems: list[Problem] = [] + # Every markdown field with the location to report it against, kept so that the + # whole set can then be compiled in one go rather than a field at a time. + fields: list[tuple[str, str]] = [] + images: list[str] = [] + + def check( + markdown: str, question: Question, location: str, compiled: bool = True + ) -> list[Problem]: + if compiled: + fields.append((location, markdown)) + return _markdown_problems(markdown, question, location) for number, question in enumerate(question_set.questions, start=1): where = f'Question {number} "{question.title}"' - problems += _markdown_problems( - question.main_text, question, f"{where}, main text" - ) + problems += check(question.main_text, question, f"{where}, main text") + images += question.images for image in question.images: if not Path(image).is_file(): problems.append(Problem(where, f"there is no image file at {image}")) @@ -72,30 +86,39 @@ def validate(question_set: Set) -> list[Problem]: ("worked solution", part.worked_solution), ("answer", part.answer), ): - problems += _markdown_problems( - markdown, question, f"{part_where}, {field}" - ) + problems += check(markdown, question, f"{part_where}, {field}") for area_number, area in enumerate(part.response_areas, start=1): area_where = f"{part_where}, answer box {area_number}" problems += [ Problem(area_where, message) for message in _area_problems(area) ] + # An answer box is only in the PDF if it is marked to be, so LaTeX it + # would not compile cannot break one unless it is. for field, markdown in ( ("pre_text", area.pre_text), ("post_text", area.post_text), ("content_after", area.content_after), ): - problems += _markdown_problems( - markdown, question, f"{area_where}, {field}" + problems += check( + markdown, + question, + f"{area_where}, {field}", + area.include_in_pdf, ) options = (area.config or {}).get("options") if isinstance(options, list): for option_number, option in enumerate(options, start=1): - problems += _markdown_problems( - option, question, f"{area_where}, option {option_number}" + problems += check( + option, + question, + f"{area_where}, option {option_number}", + area.include_in_pdf, ) + if compile: + problems += pdf.problems(fields, images) + return problems diff --git a/in2lambda/validation/pdf/__init__.py b/in2lambda/validation/pdf/__init__.py new file mode 100644 index 0000000..dfdf6f2 --- /dev/null +++ b/in2lambda/validation/pdf/__init__.py @@ -0,0 +1,199 @@ +"""Compiles a set the way Lambda Feedback makes a PDF of it, and reads the errors back. + +Lambda Feedback renders question PDFs with lambda-feedback/PDF-generator: pandoc with +``template.latex`` beside this file, then xelatex. Markdown that pipeline refuses is a +fault in the set, so the whole set is compiled once here and each LaTeX error is traced +back to the field it came from. + +The trick for tracing is a marker: the document handed to pandoc carries a raw-LaTeX +comment naming the field before each field's markdown, and pandoc copies raw blocks +through untouched. ``xelatex -file-line-error`` then reports every error as +``set.tex:: ``, and the last marker above that line names the field. + +pandoc and xelatex are both optional, as they are everywhere else in in2lambda: without +them this reports what to install rather than raising. +""" + +import re +import shutil +import subprocess +import tempfile +from pathlib import Path + +from in2lambda.api.problem import Problem + +_TEMPLATE = Path(__file__).with_name("template.latex") +"""The PDF generator's own pandoc template - see the README beside it.""" + +_TOOLS = { + "pandoc": "pandoc (see https://pandoc.org/installing.html)", + "xelatex": ( + "xelatex (apt install texlive-xetex texlive-latex-recommended" + " texlive-latex-extra texlive-science texlive-lang-chinese" + " fonts-noto-core fonts-noto-cjk)" + ), +} + +_SET = "The set" +"""Where an error that is not inside any one field is reported against.""" + +_MARKER = "% in2lambda: " + +_ERROR = re.compile( + r"^(?:\./)?(\S+\.(?:tex|sty|cls|def|cfg|fd|ltx)):(\d+): (.+)$", re.MULTILINE +) +"""One ``-file-line-error`` line. The file is only ``set.tex`` for the set's own text.""" + +_IMAGE = re.compile(r"(!\[[^\]]*\]\()([^)]*)(\))") +"""A markdown image with its path apart, so that the path can be rewritten or dropped.""" + +_TIMEOUT = 120 +"""Seconds for pandoc or xelatex. A set that takes longer is reported, not waited for.""" + + +def missing_tools() -> list[str]: + """What is needed to compile a set but is not installed, each saying how to get it. + + Returns: + One line per missing tool, or an empty list if a set can be compiled here. + """ + return [hint for tool, hint in _TOOLS.items() if shutil.which(tool) is None] + + +def problems(fields: list[tuple[str, str]], images: list[str]) -> list[Problem]: + """Everything the PDF generator's pandoc and xelatex refuse, by the field it is in. + + Args: + fields: Every markdown field of the set in the order it is written, each with + the location - ``Question 1 "Title", part (a), text`` - to report against. + images: Every image path the set's questions hold. Those that exist are put + beside the compiled document under their file name, since that is how the + export refers to them; a reference to any other is dropped, as the image + check already reports it. + + Returns: + One :class:`~in2lambda.api.problem.Problem` per distinct LaTeX error. An empty + list means the set compiles as Lambda Feedback will compile it. + """ + if missing := missing_tools(): + return [ + Problem( + _SET, + "not compiled as the PDF generator would: install " + + " and ".join(missing), + ) + ] + + try: + return _compiled(fields, images) + except subprocess.TimeoutExpired as expired: + # TeX can be made to loop forever, which is itself a fault in the set. + return [ + Problem( + _SET, + f"the PDF generator cannot compile this: {expired.cmd[0]} did not" + f" finish within {_TIMEOUT} seconds", + ) + ] + + +def _compiled(fields: list[tuple[str, str]], images: list[str]) -> list[Problem]: + """The set run through pandoc and then xelatex in a directory of its own.""" + with tempfile.TemporaryDirectory() as directory: + work = Path(directory) + available = set() + for image in images: + if Path(image).is_file(): + shutil.copy(image, work / Path(image).name) + available.add(Path(image).name) + + run = subprocess.run( + [ + "pandoc", + "-f", + "markdown-implicit_figures", + "-t", + "latex", + "-s", + f"--template={_TEMPLATE}", + "-o", + "set.tex", + ], + input=_marked_document(fields, available), + capture_output=True, + text=True, + # Not the locale's encoding: a set holding any non-ASCII character would + # then fail to even be handed over under, say, LC_ALL=C. + encoding="utf-8", + cwd=work, + timeout=_TIMEOUT, + ) + if run.returncode: + return [Problem(_SET, f"pandoc cannot read the set: {run.stderr.strip()}")] + + latex = (work / "set.tex").read_text(encoding="utf-8") + run = subprocess.run( + [ + "xelatex", + "-interaction=nonstopmode", + "-file-line-error", + "-no-shell-escape", + "set.tex", + ], + stdin=subprocess.DEVNULL, + capture_output=True, + text=True, + encoding="utf-8", + cwd=work, + timeout=_TIMEOUT, + ) + return _reported(run.stdout, _locations(latex)) + + +def _marked_document(fields: list[tuple[str, str]], available: set[str]) -> str: + """The whole set as one markdown document, each field under a marker naming it. + + The fence is four backticks so that a field which itself contains a code block + cannot close the marker's raw-LaTeX block early. + """ + blocks = [] + for location, markdown in fields: + markdown = _IMAGE.sub( + lambda image: ( + f"{image[1]}{Path(image[2]).name}{image[3]}" + if Path(image[2]).name in available + else "" + ), + markdown, + ) + blocks.append(f"````{{=latex}}\n{_MARKER}{location}\n````\n\n{markdown}\n") + return "\n".join(blocks) + + +def _locations(latex: str) -> list[tuple[int, str]]: + """Each marker in the generated LaTeX as the line it is on and the field it names.""" + return [ + (number, line.partition(_MARKER)[2]) + for number, line in enumerate(latex.splitlines(), start=1) + if line.startswith(_MARKER) + ] + + +def _reported(log: str, locations: list[tuple[int, str]]) -> list[Problem]: + """The xelatex log's errors as problems, each against the field it happened in. + + The same error repeated - a command used twice, say - is one problem, since the + author has one thing to go and fix. + """ + found = [] + for error in _ERROR.finditer(log): + file, line, message = error[1], int(error[2]), error[3].strip() + where = _SET + if file == "set.tex": + for number, location in locations: + if number <= line: + where = location + problem = Problem(where, f"the PDF generator cannot compile this: {message}") + if problem not in found: + found.append(problem) + return found diff --git a/tests/fixtures/problems/README.md b/tests/fixtures/problems/README.md index 59b2a46..59523b2 100644 --- a/tests/fixtures/problems/README.md +++ b/tests/fixtures/problems/README.md @@ -4,6 +4,10 @@ Each folder here is a hand-written Lambda Feedback export exhibiting exactly one `in2lambda.validation.validate` looks for, beside the `expected.txt` report it should produce: one `str(Problem)` line per problem, which the test compares sorted. +The reports are the ones produced with pandoc and xelatex installed, since the validator then +also compiles the set as Lambda Feedback's PDF generator does. Without them those tests are +skipped, because the report would be missing whatever the compiler would have said. + They are written by hand rather than exported by the platform, because the platform does not produce broken sets. Real exports live in `../exports`, and the same test suite checks that none of them is reported as having a problem. diff --git a/tests/fixtures/problems/latex_fault/expected.txt b/tests/fixtures/problems/latex_fault/expected.txt new file mode 100644 index 0000000..12b65a2 --- /dev/null +++ b/tests/fixtures/problems/latex_fault/expected.txt @@ -0,0 +1 @@ +Question 1 "Undefined command", part (a), text: the PDF generator cannot compile this: Undefined control sequence. diff --git a/tests/fixtures/problems/latex_fault/question_000_Undefined_command.json b/tests/fixtures/problems/latex_fault/question_000_Undefined_command.json new file mode 100644 index 0000000..1c1001b --- /dev/null +++ b/tests/fixtures/problems/latex_fault/question_000_Undefined_command.json @@ -0,0 +1,22 @@ +{ + "orderNumber": 0, + "title": "Undefined command", + "masterContent": "", + "publish": true, + "displayFinalAnswer": true, + "displayStructuredTutorial": true, + "displayWorkedSolution": true, + "displayChatbot": true, + "parts": [ + { + "orderNumber": 0, + "content": "Evaluate $x = \\nosuchcommand$.", + "answerContent": "", + "workedSolution": { + "content": "", + "children": [] + }, + "responseAreas": [] + } + ] +} diff --git a/tests/fixtures/problems/latex_fault/set_LaTeX_fault.json b/tests/fixtures/problems/latex_fault/set_LaTeX_fault.json new file mode 100644 index 0000000..9a7f5a4 --- /dev/null +++ b/tests/fixtures/problems/latex_fault/set_LaTeX_fault.json @@ -0,0 +1,7 @@ +{ + "name": "LaTeX fault", + "description": "A part whose text uses a command LaTeX does not have, so the PDF generator cannot compile it.", + "finalAnswerVisibility": "OPEN_WITH_WARNINGS", + "workedSolutionVisibility": "OPEN_WITH_WARNINGS", + "structuredTutorialVisibility": "OPEN" +} diff --git a/tests/fixtures/problems/unbalanced_maths/expected.txt b/tests/fixtures/problems/unbalanced_maths/expected.txt index b5105fc..b0e38f0 100644 --- a/tests/fixtures/problems/unbalanced_maths/expected.txt +++ b/tests/fixtures/problems/unbalanced_maths/expected.txt @@ -1 +1,2 @@ Question 1 "Continuity", part (a), worked solution: unclosed inline $ ... $ +Question 1 "Continuity", part (a), worked solution: the PDF generator cannot compile this: Missing $ inserted. diff --git a/tests/fixtures/problems/unsupported_command/expected.txt b/tests/fixtures/problems/unsupported_command/expected.txt index 0f50da1..f9799ff 100644 --- a/tests/fixtures/problems/unsupported_command/expected.txt +++ b/tests/fixtures/problems/unsupported_command/expected.txt @@ -1 +1,2 @@ Question 1 "Magnitude", part (a), text: KaTeX does not render \norm; write \mathbf instead +Question 1 "Magnitude", part (a), text: the PDF generator cannot compile this: Undefined control sequence. diff --git a/tests/test_validation.py b/tests/test_validation.py index 7b2ebcf..c37805f 100644 --- a/tests/test_validation.py +++ b/tests/test_validation.py @@ -9,16 +9,26 @@ on the ``Summer2025`` branch. """ +import os +import shutil +import subprocess +import sys from pathlib import Path import pytest from conftest import EXPORTS, PROBLEM_SETS from in2lambda.api.set import Set -from in2lambda.validation import MathDelimiterError, validate +from in2lambda.validation import MathDelimiterError, pdf, validate E = MathDelimiterError +needs_compiler = pytest.mark.skipif( + bool(pdf.missing_tools()), + reason="compiling the set as the PDF generator does needs pandoc and xelatex", +) +"""The fixtures are reported with the PDF generator's toolchain installed; CI has it.""" + VALID = [ "This is an inline math expression: $x = y$.", "This is an inline math expression: $x = y$", @@ -83,9 +93,10 @@ def _messages(markdown: str) -> list[str]: """What the validator says about a single piece of markdown.""" question_set = Set() question_set.add_question("Markdown", markdown) - return [problem.message for problem in validate(question_set)] + return [problem.message for problem in validate(question_set, compile=False)] +@needs_compiler @pytest.mark.parametrize("problem_set", PROBLEM_SETS, ids=lambda path: path.name) def test_expected_problems_are_reported(problem_set: Path) -> None: """Each hand-written export produces exactly the report written beside it.""" @@ -95,6 +106,7 @@ def test_expected_problems_are_reported(problem_set: Path) -> None: assert sorted(str(problem) for problem in found) == sorted(expected) +@needs_compiler @pytest.mark.parametrize("export", EXPORTS, ids=lambda path: path.name) def test_real_exports_have_no_problems(export: Path) -> None: """A set the platform wrote and accepted back must never be reported.""" @@ -119,6 +131,52 @@ def test_image_that_is_not_on_disk_is_reported(tmp_path: Path) -> None: question_set.add_question("Rocket", "![pictureTag](rocket.png)") question_set.current_question.images.append(str(tmp_path / "rocket.png")) - assert [problem.message for problem in question_set.problems()] == [ + assert [problem.message for problem in question_set.problems(compile=False)] == [ f"there is no image file at {tmp_path / 'rocket.png'}" ] + + +@needs_compiler +def test_non_ascii_is_compiled_whatever_the_locale() -> None: + """Exports are full of curly quotes, and containers are often not UTF-8 locales. + + Run in a process of its own because the locale is read when Python starts. + """ + script = ( + "from in2lambda.api.set import Set\n" + "from in2lambda.validation import validate\n" + "question_set = Set()\n" + "question_set.add_question('Quotes', 'The rocket\\u2019s mass.')\n" + "print(len(validate(question_set)))\n" + ) + run = subprocess.run( + [sys.executable, "-c", script], + env=os.environ + | { + "LC_ALL": "C", + "LANG": "C", + "PYTHONUTF8": "0", + "PYTHONCOERCECLOCALE": "0", + "PYTHONIOENCODING": "utf-8", + }, + capture_output=True, + text=True, + ) + + assert run.returncode == 0, run.stderr + assert run.stdout.strip() == "0" + + +def test_missing_compiler_says_what_to_install( + monkeypatch: pytest.MonkeyPatch, +) -> None: + """Without the PDF generator's toolchain, the set is not compiled but is reported.""" + monkeypatch.setattr(shutil, "which", lambda tool: None) + question_set = Set() + question_set.add_question("Angles", "Turn through 90°.") + + problems = validate(question_set) + + assert len(problems) == 1 + assert "xelatex" in problems[0].message + assert "texlive-xetex" in problems[0].message