Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
9 changes: 9 additions & 0 deletions .github/workflows/test.yml
Original file line number Diff line number Diff line change
Expand Up @@ -22,6 +22,15 @@ jobs:
poetry install --with dev --all-extras
- name: Install Pandoc # apt version seems too old
uses: r-lib/actions/setup-pandoc@v2
# The validator compiles a set the way lambda-feedback/PDF-generator does, so it
# needs what that image installs: xelatex with braket, cancel, xeCJK and the
# Noto Sans fonts the template sets as its main and CJK fonts.
- name: Install XeLaTeX
run: |
sudo apt-get update
sudo apt-get install -y --no-install-recommends \
texlive-xetex texlive-latex-recommended texlive-latex-extra \
texlive-science texlive-lang-chinese fonts-noto-core fonts-noto-cjk
- name: Linting Checks
run: |
poetry run black --check .
Expand Down
2 changes: 2 additions & 0 deletions docs/source/quickstart.md
Original file line number Diff line number Diff line change
Expand Up @@ -75,6 +75,8 @@ By default, this generates an `out` directory in the same place that the command

Before writing anything, in2lambda prints the problems it can detect that would stop the set importing or make it render wrongly — an answer that doesn't fit the box marking it, a figure the export won't contain, maths that KaTeX can't display. Each names the question, part and field to go and look at. They are warnings rather than errors: the `out` directory is written either way, since a problem found here may well be deliberate.

With [xelatex](https://tug.org/texlive/) installed alongside pandoc, the set is also compiled the way Lambda Feedback makes a PDF of it, and any LaTeX error names the field it is in. Without it, one warning says which packages to install instead.

Check the [command line tool reference](reference/command-line) for more information.

## 3. Import into Lambda Feedback
Expand Down
10 changes: 7 additions & 3 deletions in2lambda/api/set.py
Original file line number Diff line number Diff line change
Expand Up @@ -103,12 +103,16 @@ def increment_current_question(self) -> None:
"""
self._current_question_index += 1

def problems(self) -> list[Problem]:
def problems(self, compile: bool = True) -> list[Problem]:
r"""Everything in2lambda can tell Lambda Feedback would refuse or render wrongly.

This is a report, not a refusal: the set can still be written out, since a
problem found here may well be deliberate.

Args:
compile: Whether to also compile the set as Lambda Feedback's PDF generator
will, which needs pandoc and xelatex installed.

Returns:
One :class:`~in2lambda.api.problem.Problem` per problem found, each naming
the question, part and field to go and look at.
Expand All @@ -118,14 +122,14 @@ def problems(self) -> list[Problem]:
>>> s = Set()
>>> s.add_question("Momentum", "The rocket is at $45^\\circ$.")
>>> s.current_question.images.append("no_such_file.png")
>>> for problem in s.problems():
>>> for problem in s.problems(compile=False):
... print(problem)
Question 1 "Momentum", main text: ^\circ does not display; write the degree sign ° instead
Question 1 "Momentum": there is no image file at no_such_file.png
"""
from in2lambda.validation import validate

return validate(self)
return validate(self, compile)

def to_json(self, output_dir: str) -> None:
"""Turns this set into Lambda Feedback JSON/ZIP files.
Expand Down
47 changes: 35 additions & 12 deletions in2lambda/validation/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -18,6 +18,7 @@
from in2lambda.api.response_area import ResponseArea
from in2lambda.api.set import Set
from in2lambda.katex_convert.katex_convert import unsupported_commands
from in2lambda.validation import pdf
from in2lambda.validation.delimiters import MathDelimiterError, math_delimiter_checker

__all__ = ["MathDelimiterError", "Problem", "math_delimiter_checker", "validate"]
Expand All @@ -34,11 +35,14 @@
"""``^\\circ``, with or without braces around it."""


def validate(question_set: Set) -> list[Problem]:
def validate(question_set: Set, compile: bool = True) -> list[Problem]:
r"""Everything in2lambda can tell is wrong with a set, in the order it is written.

Args:
question_set: The set about to be exported.
compile: Whether to also compile the set as Lambda Feedback's PDF generator
will, which needs pandoc and xelatex - see
:mod:`in2lambda.validation.pdf`.

Returns:
One :class:`~in2lambda.api.problem.Problem` per problem found, each naming the
Expand All @@ -50,17 +54,27 @@ def validate(question_set: Set) -> list[Problem]:
>>> from in2lambda.validation import validate
>>> s = Set()
>>> s.add_question("Angles", "Turn through $90^\\circ$.")
>>> [str(problem) for problem in validate(s)]
>>> [str(problem) for problem in validate(s, compile=False)]
['Question 1 "Angles", main text: ^\\circ does not display; write the degree sign ° instead']
"""
problems: list[Problem] = []
# Every markdown field with the location to report it against, kept so that the
# whole set can then be compiled in one go rather than a field at a time.
fields: list[tuple[str, str]] = []
images: list[str] = []

def check(
markdown: str, question: Question, location: str, compiled: bool = True
) -> list[Problem]:
if compiled:
fields.append((location, markdown))
return _markdown_problems(markdown, question, location)

for number, question in enumerate(question_set.questions, start=1):
where = f'Question {number} "{question.title}"'
problems += _markdown_problems(
question.main_text, question, f"{where}, main text"
)
problems += check(question.main_text, question, f"{where}, main text")

images += question.images
for image in question.images:
if not Path(image).is_file():
problems.append(Problem(where, f"there is no image file at {image}"))
Expand All @@ -72,30 +86,39 @@ def validate(question_set: Set) -> list[Problem]:
("worked solution", part.worked_solution),
("answer", part.answer),
):
problems += _markdown_problems(
markdown, question, f"{part_where}, {field}"
)
problems += check(markdown, question, f"{part_where}, {field}")

for area_number, area in enumerate(part.response_areas, start=1):
area_where = f"{part_where}, answer box {area_number}"
problems += [
Problem(area_where, message) for message in _area_problems(area)
]
# An answer box is only in the PDF if it is marked to be, so LaTeX it
# would not compile cannot break one unless it is.
for field, markdown in (
("pre_text", area.pre_text),
("post_text", area.post_text),
("content_after", area.content_after),
):
problems += _markdown_problems(
markdown, question, f"{area_where}, {field}"
problems += check(
markdown,
question,
f"{area_where}, {field}",
area.include_in_pdf,
)
options = (area.config or {}).get("options")
if isinstance(options, list):
for option_number, option in enumerate(options, start=1):
problems += _markdown_problems(
option, question, f"{area_where}, option {option_number}"
problems += check(
option,
question,
f"{area_where}, option {option_number}",
area.include_in_pdf,
)

if compile:
problems += pdf.problems(fields, images)

return problems


Expand Down
199 changes: 199 additions & 0 deletions in2lambda/validation/pdf/__init__.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,199 @@
"""Compiles a set the way Lambda Feedback makes a PDF of it, and reads the errors back.

Lambda Feedback renders question PDFs with lambda-feedback/PDF-generator: pandoc with
``template.latex`` beside this file, then xelatex. Markdown that pipeline refuses is a
fault in the set, so the whole set is compiled once here and each LaTeX error is traced
back to the field it came from.

The trick for tracing is a marker: the document handed to pandoc carries a raw-LaTeX
comment naming the field before each field's markdown, and pandoc copies raw blocks
through untouched. ``xelatex -file-line-error`` then reports every error as
``set.tex:<line>: <message>``, and the last marker above that line names the field.

pandoc and xelatex are both optional, as they are everywhere else in in2lambda: without
them this reports what to install rather than raising.
"""

import re
import shutil
import subprocess
import tempfile
from pathlib import Path

from in2lambda.api.problem import Problem

_TEMPLATE = Path(__file__).with_name("template.latex")
"""The PDF generator's own pandoc template - see the README beside it."""

_TOOLS = {
"pandoc": "pandoc (see https://pandoc.org/installing.html)",
"xelatex": (
"xelatex (apt install texlive-xetex texlive-latex-recommended"
" texlive-latex-extra texlive-science texlive-lang-chinese"
" fonts-noto-core fonts-noto-cjk)"
),
}

_SET = "The set"
"""Where an error that is not inside any one field is reported against."""

_MARKER = "% in2lambda: "

_ERROR = re.compile(
r"^(?:\./)?(\S+\.(?:tex|sty|cls|def|cfg|fd|ltx)):(\d+): (.+)$", re.MULTILINE
)
"""One ``-file-line-error`` line. The file is only ``set.tex`` for the set's own text."""

_IMAGE = re.compile(r"(!\[[^\]]*\]\()([^)]*)(\))")
"""A markdown image with its path apart, so that the path can be rewritten or dropped."""

_TIMEOUT = 120
"""Seconds for pandoc or xelatex. A set that takes longer is reported, not waited for."""


def missing_tools() -> list[str]:
"""What is needed to compile a set but is not installed, each saying how to get it.

Returns:
One line per missing tool, or an empty list if a set can be compiled here.
"""
return [hint for tool, hint in _TOOLS.items() if shutil.which(tool) is None]


def problems(fields: list[tuple[str, str]], images: list[str]) -> list[Problem]:
"""Everything the PDF generator's pandoc and xelatex refuse, by the field it is in.

Args:
fields: Every markdown field of the set in the order it is written, each with
the location - ``Question 1 "Title", part (a), text`` - to report against.
images: Every image path the set's questions hold. Those that exist are put
beside the compiled document under their file name, since that is how the
export refers to them; a reference to any other is dropped, as the image
check already reports it.

Returns:
One :class:`~in2lambda.api.problem.Problem` per distinct LaTeX error. An empty
list means the set compiles as Lambda Feedback will compile it.
"""
if missing := missing_tools():
return [
Problem(
_SET,
"not compiled as the PDF generator would: install "
+ " and ".join(missing),
)
]

try:
return _compiled(fields, images)
except subprocess.TimeoutExpired as expired:
# TeX can be made to loop forever, which is itself a fault in the set.
return [
Problem(
_SET,
f"the PDF generator cannot compile this: {expired.cmd[0]} did not"
f" finish within {_TIMEOUT} seconds",
)
]


def _compiled(fields: list[tuple[str, str]], images: list[str]) -> list[Problem]:
"""The set run through pandoc and then xelatex in a directory of its own."""
with tempfile.TemporaryDirectory() as directory:
work = Path(directory)
available = set()
for image in images:
if Path(image).is_file():
shutil.copy(image, work / Path(image).name)
available.add(Path(image).name)

run = subprocess.run(
[
"pandoc",
"-f",
"markdown-implicit_figures",
"-t",
"latex",
"-s",
f"--template={_TEMPLATE}",
"-o",
"set.tex",
],
input=_marked_document(fields, available),
capture_output=True,
text=True,
# Not the locale's encoding: a set holding any non-ASCII character would
# then fail to even be handed over under, say, LC_ALL=C.
encoding="utf-8",
cwd=work,
timeout=_TIMEOUT,
)
if run.returncode:
return [Problem(_SET, f"pandoc cannot read the set: {run.stderr.strip()}")]

latex = (work / "set.tex").read_text(encoding="utf-8")
run = subprocess.run(
[
"xelatex",
"-interaction=nonstopmode",
"-file-line-error",
"-no-shell-escape",
"set.tex",
],
stdin=subprocess.DEVNULL,
capture_output=True,
text=True,
encoding="utf-8",
cwd=work,
timeout=_TIMEOUT,
)
return _reported(run.stdout, _locations(latex))


def _marked_document(fields: list[tuple[str, str]], available: set[str]) -> str:
"""The whole set as one markdown document, each field under a marker naming it.

The fence is four backticks so that a field which itself contains a code block
cannot close the marker's raw-LaTeX block early.
"""
blocks = []
for location, markdown in fields:
markdown = _IMAGE.sub(
lambda image: (
f"{image[1]}{Path(image[2]).name}{image[3]}"
if Path(image[2]).name in available
else ""
),
markdown,
)
blocks.append(f"````{{=latex}}\n{_MARKER}{location}\n````\n\n{markdown}\n")
return "\n".join(blocks)


def _locations(latex: str) -> list[tuple[int, str]]:
"""Each marker in the generated LaTeX as the line it is on and the field it names."""
return [
(number, line.partition(_MARKER)[2])
for number, line in enumerate(latex.splitlines(), start=1)
if line.startswith(_MARKER)
]


def _reported(log: str, locations: list[tuple[int, str]]) -> list[Problem]:
"""The xelatex log's errors as problems, each against the field it happened in.

The same error repeated - a command used twice, say - is one problem, since the
author has one thing to go and fix.
"""
found = []
for error in _ERROR.finditer(log):
file, line, message = error[1], int(error[2]), error[3].strip()
where = _SET
if file == "set.tex":
for number, location in locations:
if number <= line:
where = location
problem = Problem(where, f"the PDF generator cannot compile this: {message}")
if problem not in found:
found.append(problem)
return found
4 changes: 4 additions & 0 deletions tests/fixtures/problems/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -4,6 +4,10 @@ Each folder here is a hand-written Lambda Feedback export exhibiting exactly one
`in2lambda.validation.validate` looks for, beside the `expected.txt` report it should produce:
one `str(Problem)` line per problem, which the test compares sorted.

The reports are the ones produced with pandoc and xelatex installed, since the validator then
also compiles the set as Lambda Feedback's PDF generator does. Without them those tests are
skipped, because the report would be missing whatever the compiler would have said.

They are written by hand rather than exported by the platform, because the platform does not
produce broken sets. Real exports live in `../exports`, and the same test suite checks that none
of them is reported as having a problem.
Expand Down
1 change: 1 addition & 0 deletions tests/fixtures/problems/latex_fault/expected.txt
Original file line number Diff line number Diff line change
@@ -0,0 +1 @@
Question 1 "Undefined command", part (a), text: the PDF generator cannot compile this: Undefined control sequence.
Original file line number Diff line number Diff line change
@@ -0,0 +1,22 @@
{
"orderNumber": 0,
"title": "Undefined command",
"masterContent": "",
"publish": true,
"displayFinalAnswer": true,
"displayStructuredTutorial": true,
"displayWorkedSolution": true,
"displayChatbot": true,
"parts": [
{
"orderNumber": 0,
"content": "Evaluate $x = \\nosuchcommand$.",
"answerContent": "",
"workedSolution": {
"content": "",
"children": []
},
"responseAreas": []
}
]
}
Loading
Loading