From 114be08f34a644a59fc03f858b36807e06f55705 Mon Sep 17 00:00:00 2001 From: "Peter B. Johnson" Date: Wed, 23 Sep 2026 22:46:13 +0100 Subject: [PATCH 1/6] implement: Fold notation in in2lambda's comparison (t59) --- CHANGELOG.md | 2 +- in2lambda/compare.py | 71 +++++++++++++++++-- in2lambda/main.py | 11 +-- .../against_convert/PartsOneSol/differs.txt | 2 - tests/fixtures/against_convert/README.md | 18 +++-- tests/test_compare.py | 53 ++++++++++++-- 6 files changed, 135 insertions(+), 22 deletions(-) delete mode 100644 tests/fixtures/against_convert/PartsOneSol/differs.txt diff --git a/CHANGELOG.md b/CHANGELOG.md index 3651034..013e2b2 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -25,5 +25,5 @@ - `in2lambda convert FILE PartsOneSol` now exports the worked solution a document writes in a `solution` environment. Pandoc writes that environment as a Div whose classes hold `solution`, and the filter recognised only a Div whose first block reads `Solution`, so a document using the environment exported every question with an empty worked solution. A Div whose first block reads `Solution` is still recognised. - Importing `in2lambda.katex_convert` no longer writes a file called `log` into the working directory. That module reports what it changed in an expression to the `in2lambda.katex_convert` logger, which is silent unless the application configures logging. - `in2lambda convert` now reads a .docx that holds an image. in2lambda looks in the document for the directories a `\graphicspath` names, and read the document as UTF-8 text to find them. A .docx is a zip file, so converting a Word document holding a figure raised `UnicodeDecodeError`. in2lambda now reads a document that is not UTF-8 text as naming no directory, which is what a .docx names. -- `in2lambda compare BUILT_ZIP EXPORT_DIR` compares two Lambda Feedback sets, each given as a folder or a zip, so that a set in2lambda wrote can be checked against the export it should reproduce. Each question's main text is compared, and each part's text and worked solution, and every difference is printed naming the question, the part and the field. Three differences in wording are taken off both sides first: a run of whitespace is compared as one space, an image is compared by the file's name, and a part holding neither text nor a worked solution is dropped where it is the question's only part. `--known FILE` names the differences the two sets are known to have, one line per difference with the ticket that would close it written after ` # `, and `in2lambda compare` exits 1 where the differences found are not the differences that file names. `in2lambda.compare.differences` and `in2lambda.compare.known` are the two functions behind the command. +- `in2lambda compare BUILT_ZIP EXPORT_DIR` compares two Lambda Feedback sets, each given as a folder or a zip, so that a set in2lambda wrote can be checked against the export it should reproduce. Each question's main text is compared, and each part's text and worked solution, and every difference is printed naming the question, the part and the field. The differences in wording that are not differences in what a question says are taken off both sides first: a run of whitespace is compared as one space, a line holding nothing but hyphens is dropped, each curly quote is compared as the straight quote, an image is compared by the file's name, and a part holding neither text nor a worked solution is dropped where it is the question's only part. Inside every `$ ... $` and `$$ ... $$`, `\left` and `\right` are removed, `~`, `\,` and `\space` are compared as a space, and every run of whitespace is dropped, except that a run between a control word and a following letter is compared as one space, so that `$z=2+3 i$` and `$z=2+3i$` are the same expression and `$\alpha x$` and `$\alphax$` are two. `--known FILE` names the differences the two sets are known to have, one line per difference with the ticket that would close it written after ` # `, and `in2lambda compare` exits 1 where the differences found are not the differences that file names. `in2lambda.compare.differences` and `in2lambda.compare.known` are the two functions behind the command. - The rest of the Python API is unchanged: `in2lambda.main.runner` and everything under `in2lambda.api` take the same arguments and return the same objects. diff --git a/in2lambda/compare.py b/in2lambda/compare.py index 439ce30..5a162a5 100644 --- a/in2lambda/compare.py +++ b/in2lambda/compare.py @@ -1,4 +1,4 @@ -"""Compares two sets question by question, naming every place they say something else. +r"""Compares two sets question by question, naming every place they say something else. `in2lambda convert` writes a set, `in2lambda build` writes a set from a draft, and Lambda Feedback exports a set. :func:`differences` compares any two of them in question and part @@ -6,12 +6,25 @@ one line per difference, naming the question, the part and the field as `in2lambda.validation` names them. -Three differences in wording are not differences in what a question says, and are taken +These differences in wording are not differences in what a question says, and are taken off both sides before comparing: - **Whitespace.** Every run of whitespace is compared as one space, because a draft quotes the lines pandoc wrapped where `in2lambda convert` writes a paragraph on one line. +- **Separator lines.** A line holding nothing but three or more hyphens is dropped, + because Lambda Feedback writes one around a display maths block where a document + writes nothing. +- **Quotes.** ``‘`` and ``’`` are compared as ``'``, and ``“`` and ``”`` as ``"``, + because pandoc's LaTeX reader writes the curly quote where its commonmark_x writer + writes the straight one. +- **Maths notation.** Inside every ``$ ... $`` and ``$$ ... $$``, ``\left`` and + ``\right`` are removed, ``~``, ``\,`` and ``\space`` are compared as a space, and + every run of whitespace is dropped, except that a run between a control word and a + following letter is compared as one space. LaTeX renders ``$z=2+3 i$`` and + ``$z=2+3i$`` the same, and ``\mathrm{~m}`` and ``\mathrm{m}`` the same, where + ``\alpha x`` and ``\alphax`` are two different expressions. An export writes + ``$z = 2+3 i$`` where `in2lambda convert` writes ``$z=2+3i$``. - **Image references.** An image is compared by the file's name, because `in2lambda convert` writes every image as ``![pictureTag](path)`` where a draft keeps the alt text the document wrote, and an export names each file as ``media/`` holds it @@ -25,6 +38,7 @@ after `` # ``. """ +import re from itertools import zip_longest from pathlib import Path from typing import Any, Optional @@ -32,19 +46,64 @@ from in2lambda.api.question import Question from in2lambda.api.set import Set from in2lambda.json_convert.json_convert import _IMAGE -from in2lambda.validation import _location +from in2lambda.validation import _COMMAND, _MATHS, _location _TICKET = " # " """What a line of a differs.txt names the ticket closing it after.""" +_RULE = re.compile(r"(?m)^[ \t]*-{3,}[ \t]*$") +"""A line holding nothing but hyphens, which is the separator Lambda Feedback writes.""" + +_QUOTES = str.maketrans({"‘": "'", "’": "'", "“": '"', "”": '"'}) +"""Each curly quote and the straight quote it is compared as.""" + +_SIZE = re.compile(r"\\(?:left|right)(?![a-zA-Z])") +r"""``\left`` and ``\right``, which size a delimiter without changing which it is.""" + +_LATEX_SPACE = re.compile(r"~|\\,|\\space(?![a-zA-Z])") +"""The three ways of writing a space inside maths.""" + +_SPACING = re.compile(rf"({_COMMAND.pattern})\s+(?=[a-zA-Z])|\s+") +"""A run of whitespace inside maths, with the control word it ends where one precedes it +and a letter follows it.""" + + +def _spacing(whitespace: re.Match[str]) -> str: + r"""One space where a run of whitespace ends a control word, and nothing elsewhere. + + ``\alpha x`` is two symbols and ``\alphax`` is a control word nothing defines, so + the space between a control word and a letter is the only whitespace LaTeX renders. + """ + return f"{whitespace[1]} " if whitespace[1] else "" + + +def _maths(expression: re.Match[str]) -> str: + """One ``$ ... $`` or ``$$ ... $$`` with the notation that is not the maths folded. + + Args: + expression: A match of `in2lambda.validation._MATHS`, holding the display maths + it found in its first group and the inline maths in its second. + """ + display = expression[1] is not None + tex = expression[1] if display else expression[2] + delimiter = "$$" if display else "$" + folded = _LATEX_SPACE.sub(" ", _SIZE.sub("", tex)) + return f"{delimiter}{_SPACING.sub(_spacing, folded)}{delimiter}" + def _text(markdown: str) -> str: """A field with the differences in wording that are not differences taken off. - Every run of whitespace becomes one space, and every image reference is written as - the file's name alone. The module docstring says why. + A separator line is dropped, each curly quote becomes a straight quote, the maths + notation inside every ``$ ... $`` and ``$$ ... $$`` is folded, every image reference + is written as the file's name alone, and every run of whitespace becomes one space. + The module docstring says why. """ - named = _IMAGE.sub(lambda reference: f"![]({Path(reference[1]).name})", markdown) + # Before the whitespace collapse below, which writes the field on one line and + # leaves no line for _RULE to match. + without_rules = _RULE.sub("", markdown) + folded = _MATHS.sub(_maths, without_rules.translate(_QUOTES)) + named = _IMAGE.sub(lambda reference: f"![]({Path(reference[1]).name})", folded) return " ".join(named.split()) diff --git a/in2lambda/main.py b/in2lambda/main.py index af77230..95b2bd3 100644 --- a/in2lambda/main.py +++ b/in2lambda/main.py @@ -638,11 +638,14 @@ def compare(built_zip: str, export_dir: str, known_path: Optional[str]) -> None: Each argument is a Lambda Feedback set, as a folder or as a zip. Each question's main text is compared, and each part's text and worked solution, and every difference is - printed naming the question, the part and the field. Three differences in wording are - taken off both sides first: a run of whitespace is compared as one space, an image is + printed naming the question, the part and the field. The differences in wording that + are not differences in what a question says are taken off both sides first: a run of + whitespace is compared as one space, a line of hyphens is dropped, a curly quote is + compared as a straight quote, the notation inside maths is folded, an image is compared by the file's name, and a lone empty part is dropped. in2lambda.compare says - why. --known names a file of the differences the two sets are known to have, one per - line as this command prints it, with a ticket written after " # ". in2lambda compare + which notation and why. --known names a file of the differences the two sets are + known to have, one per line as this command prints it, with a ticket written after + " # ". in2lambda compare exits 1 where the differences found are not the differences --known names. """ with _message_not_traceback(): diff --git a/tests/fixtures/against_convert/PartsOneSol/differs.txt b/tests/fixtures/against_convert/PartsOneSol/differs.txt deleted file mode 100644 index c87077b..0000000 --- a/tests/fixtures/against_convert/PartsOneSol/differs.txt +++ /dev/null @@ -1,2 +0,0 @@ -Question 1 "", part (b), text: the draft says "The filter still works even if there aren't any parts" and convert says 'The filter still works even if there aren’t any parts' # smart quotes: pandoc's LaTeX reader writes ’ where its commonmark_x writer writes ' -Question 2 "", part (a), worked solution: the draft says "And here's the solution" and convert says 'And here’s the solution' # smart quotes: pandoc's LaTeX reader writes ’ where its commonmark_x writer writes ' diff --git a/tests/fixtures/against_convert/README.md b/tests/fixtures/against_convert/README.md index a202bba..c525818 100644 --- a/tests/fixtures/against_convert/README.md +++ b/tests/fixtures/against_convert/README.md @@ -26,12 +26,24 @@ document itself, as the ranges in `fixtures/sources` are. They were produced wit The comparison is `in2lambda.compare.differences`, whose docstring states these rules, so that this README and the code do not drift apart. -Each question's main text is compared, and each part's text and worked solution. Three +Each question's main text is compared, and each part's text and worked solution. These differences between the routes are not differences in what a question says, and are taken off both sides before comparing: - **Line breaks.** The draft quotes the lines pandoc wrapped; convert writes a paragraph on one line. Every run of whitespace is compared as one space. +- **Separator lines.** A line holding nothing but three or more hyphens is dropped, + because Lambda Feedback writes one around a display maths block where a document + writes nothing. +- **Quotes.** Pandoc's LaTeX reader writes `’` where its `commonmark_x` writer writes + `'`, so `aren’t` and `aren't` are the same wording. Each curly quote is compared as + the straight quote. +- **Maths notation.** Inside every `$ ... $` and `$$ ... $$`, `\left` and `\right` are + removed, `~`, `\,` and `\space` are compared as a space, and every run of whitespace + is dropped, except that a run between a control word and a following letter is + compared as one space. LaTeX renders `$z=2+3 i$` and `$z=2+3i$` the same, and + `\mathrm{~m}` and `\mathrm{m}` the same, where `\alpha x` and `\alphax` are two + different expressions. - **Image references.** Convert writes every image as `![pictureTag](path)`; the draft keeps the alt text the document wrote, which is empty for `\includegraphics`. The set read back from the zip names each file as it sits in the export's `media/`, where the set convert @@ -50,7 +62,3 @@ A folder with no such file is a document the two routes say the same thing about Finding a difference the file does not list fails the test, and so does agreeing where it lists one: closing a ticket below means deleting the lines it names. - -- **Smart quotes** - pandoc's LaTeX reader writes `’` where its `commonmark_x` writer - writes `'`, so convert uploads `aren’t` for `PartsOneSol/example.tex` and the draft - uploads `aren't`. diff --git a/tests/test_compare.py b/tests/test_compare.py index aaf8dbb..d220747 100644 --- a/tests/test_compare.py +++ b/tests/test_compare.py @@ -37,6 +37,51 @@ def test_a_run_of_whitespace_is_one_space() -> None: ) +def test_a_separator_line_is_dropped() -> None: + """Lambda Feedback writes a `---` line where a document writes nothing.""" + assert ( + differences( + _set("The mass is\n\n---\n\n$$m = 1$$"), _set("The mass is\n$$m=1$$") + ) + == [] + ) + + +def test_a_curly_quote_is_a_straight_quote() -> None: + """Pandoc's LaTeX reader writes `’` where its commonmark_x writer writes `'`.""" + assert differences(_set("It isn’t large."), _set("It isn't large.")) == [] + assert differences(_set("The “load”."), _set('The "load".')) == [] + + +def test_whitespace_inside_maths_is_dropped() -> None: + """LaTeX renders `$z=2+3 i$` and `$z=2+3i$` the same.""" + assert differences(_set("$z = 2+3 i$"), _set("$z=2+3i$")) == [] + assert differences(_set("$$\nF = pA\n$$"), _set("$$F=pA$$")) == [] + + +def test_a_space_between_a_control_word_and_a_letter_is_kept() -> None: + r"""`\alpha x` is two symbols and `\alphax` is a control word nothing defines.""" + assert differences(_set(r"$\alpha x$"), _set(r"$\alpha x$")) == [] + + assert differences(_set(r"$\alpha x$"), _set(r"$\alphax$")) == [ + 'Question 1 "", main text: the draft says ' + r"'$\\alpha x$' and convert says '$\\alphax$'" + ] + + +def test_a_sized_delimiter_is_the_delimiter() -> None: + r"""`\left(` and `(` render the same bracket.""" + assert differences(_set(r"$\left( x+1 \right)$"), _set("$(x+1)$")) == [] + + +def test_each_latex_space_is_a_space() -> None: + r"""`~`, `\,` and `\space` are the three ways of writing a space inside maths.""" + assert differences(_set("$a~b$"), _set("$ab$")) == [] + assert differences(_set(r"$a\,b$"), _set("$ab$")) == [] + assert differences(_set(r"$a\space b$"), _set("$ab$")) == [] + assert differences(_set(r"$5\mathrm{~m}$"), _set(r"$5\mathrm{m}$")) == [] + + def test_an_image_is_compared_by_the_file_name() -> None: """The alt text and the directory differ between the routes; the file name does not.""" assert ( @@ -67,12 +112,12 @@ def test_a_difference_names_the_question_the_part_and_the_field() -> None: expected = _set("Find the load.", ("State the pressure.", "$F = 2pA$")) assert differences(built, expected) == [ - "Question 1 \"\", part (a), worked solution: the draft says '$F = pA$' and " - "convert says '$F = 2pA$'" + "Question 1 \"\", part (a), worked solution: the draft says '$F=pA$' and " + "convert says '$F=2pA$'" ] assert differences(built, expected, "set.zip", "the export") == [ - "Question 1 \"\", part (a), worked solution: set.zip says '$F = pA$' and " - "the export says '$F = 2pA$'" + "Question 1 \"\", part (a), worked solution: set.zip says '$F=pA$' and " + "the export says '$F=2pA$'" ] From 74091de7125711d0d220ffac841d50473a5f29ae Mon Sep 17 00:00:00 2001 From: "Peter B. Johnson" Date: Wed, 23 Sep 2026 23:05:27 +0100 Subject: [PATCH 2/6] implement: Fold notation in in2lambda's comparison (t59) --- CHANGELOG.md | 2 +- in2lambda/compare.py | 33 +++++++++++++++++++----- in2lambda/main.py | 13 +++++----- tests/fixtures/against_convert/README.md | 5 ++++ tests/test_compare.py | 22 ++++++++++++++++ 5 files changed, 61 insertions(+), 14 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 013e2b2..f5e6cbc 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -25,5 +25,5 @@ - `in2lambda convert FILE PartsOneSol` now exports the worked solution a document writes in a `solution` environment. Pandoc writes that environment as a Div whose classes hold `solution`, and the filter recognised only a Div whose first block reads `Solution`, so a document using the environment exported every question with an empty worked solution. A Div whose first block reads `Solution` is still recognised. - Importing `in2lambda.katex_convert` no longer writes a file called `log` into the working directory. That module reports what it changed in an expression to the `in2lambda.katex_convert` logger, which is silent unless the application configures logging. - `in2lambda convert` now reads a .docx that holds an image. in2lambda looks in the document for the directories a `\graphicspath` names, and read the document as UTF-8 text to find them. A .docx is a zip file, so converting a Word document holding a figure raised `UnicodeDecodeError`. in2lambda now reads a document that is not UTF-8 text as naming no directory, which is what a .docx names. -- `in2lambda compare BUILT_ZIP EXPORT_DIR` compares two Lambda Feedback sets, each given as a folder or a zip, so that a set in2lambda wrote can be checked against the export it should reproduce. Each question's main text is compared, and each part's text and worked solution, and every difference is printed naming the question, the part and the field. The differences in wording that are not differences in what a question says are taken off both sides first: a run of whitespace is compared as one space, a line holding nothing but hyphens is dropped, each curly quote is compared as the straight quote, an image is compared by the file's name, and a part holding neither text nor a worked solution is dropped where it is the question's only part. Inside every `$ ... $` and `$$ ... $$`, `\left` and `\right` are removed, `~`, `\,` and `\space` are compared as a space, and every run of whitespace is dropped, except that a run between a control word and a following letter is compared as one space, so that `$z=2+3 i$` and `$z=2+3i$` are the same expression and `$\alpha x$` and `$\alphax$` are two. `--known FILE` names the differences the two sets are known to have, one line per difference with the ticket that would close it written after ` # `, and `in2lambda compare` exits 1 where the differences found are not the differences that file names. `in2lambda.compare.differences` and `in2lambda.compare.known` are the two functions behind the command. +- `in2lambda compare BUILT_ZIP EXPORT_DIR` compares two Lambda Feedback sets, each given as a folder or a zip, so that a set in2lambda wrote can be checked against the export it should reproduce. Each question's main text is compared, and each part's text and worked solution, and every difference is printed naming the question, the part and the field. The differences in wording that are not differences in what a question says are taken off both sides first: a run of whitespace is compared as one space, a line holding nothing but hyphens is dropped, each curly quote is compared as the straight quote, ` ` and ` ` are compared as a space, an image is compared by the file's name, and a part holding neither text nor a worked solution is dropped where it is the question's only part. The whitespace before and after each maths span is dropped, so that `equation of $y'+y=0$:` and `equation of$y'+y=0$:` are the same wording. Inside every `$ ... $` and `$$ ... $$`, `\left` and `\right` are removed, `~`, `\,` and `\space` are compared as a space, and every run of whitespace is dropped, except that a run between a control word and a following letter is compared as one space, so that `$z=2+3 i$` and `$z=2+3i$` are the same expression and `$\alpha x$` and `$\alphax$` are two. `--known FILE` names the differences the two sets are known to have, one line per difference with the ticket that would close it written after ` # `, and `in2lambda compare` exits 1 where the differences found are not the differences that file names. `in2lambda.compare.differences` and `in2lambda.compare.known` are the two functions behind the command. - The rest of the Python API is unchanged: `in2lambda.main.runner` and everything under `in2lambda.api` take the same arguments and return the same objects. diff --git a/in2lambda/compare.py b/in2lambda/compare.py index 5a162a5..b8b8862 100644 --- a/in2lambda/compare.py +++ b/in2lambda/compare.py @@ -18,6 +18,12 @@ - **Quotes.** ``‘`` and ``’`` are compared as ``'``, and ``“`` and ``”`` as ``"``, because pandoc's LaTeX reader writes the curly quote where its commonmark_x writer writes the straight one. +- **HTML entities for a space.** `` `` and `` `` are compared as a space, + because an export writes ``$y=0$ (Use $A$`` where `in2lambda convert` + writes ``$y=0$ (Use $A$``. +- **Whitespace beside maths.** The whitespace before and after each ``$ ... $`` and + ``$$ ... $$`` is dropped, because an export writes ``equation of $y'+y=0$:`` where + `in2lambda convert` writes ``equation of$y'+y=0$:``. - **Maths notation.** Inside every ``$ ... $`` and ``$$ ... $$``, ``\left`` and ``\right`` are removed, ``~``, ``\,`` and ``\space`` are compared as a space, and every run of whitespace is dropped, except that a run between a control word and a @@ -57,6 +63,14 @@ _QUOTES = str.maketrans({"‘": "'", "’": "'", "“": '"', "”": '"'}) """Each curly quote and the straight quote it is compared as.""" +_ENTITY = re.compile(r" | ") +"""The two HTML entities Lambda Feedback writes a space as.""" + +_SPACED_MATHS = re.compile(rf"\s*(?:{_MATHS.pattern})\s*", re.DOTALL) +"""One maths span with the whitespace touching it. Wrapping `in2lambda.validation._MATHS` +in a group that captures nothing keeps the display maths it found in the first group and +the inline maths in the second.""" + _SIZE = re.compile(r"\\(?:left|right)(?![a-zA-Z])") r"""``\left`` and ``\right``, which size a delimiter without changing which it is.""" @@ -80,9 +94,11 @@ def _spacing(whitespace: re.Match[str]) -> str: def _maths(expression: re.Match[str]) -> str: """One ``$ ... $`` or ``$$ ... $$`` with the notation that is not the maths folded. + The whitespace `_SPACED_MATHS` matched around the span is dropped with it. + Args: - expression: A match of `in2lambda.validation._MATHS`, holding the display maths - it found in its first group and the inline maths in its second. + expression: A match of `_SPACED_MATHS`, holding the display maths it found in + its first group and the inline maths in its second. """ display = expression[1] is not None tex = expression[1] if display else expression[2] @@ -94,15 +110,18 @@ def _maths(expression: re.Match[str]) -> str: def _text(markdown: str) -> str: """A field with the differences in wording that are not differences taken off. - A separator line is dropped, each curly quote becomes a straight quote, the maths - notation inside every ``$ ... $`` and ``$$ ... $$`` is folded, every image reference - is written as the file's name alone, and every run of whitespace becomes one space. - The module docstring says why. + A separator line is dropped, each HTML entity for a space becomes a space, each + curly quote becomes a straight quote, the maths notation inside every ``$ ... $`` + and ``$$ ... $$`` is folded with the whitespace touching the span, every image + reference is written as the file's name alone, and every run of whitespace becomes + one space. The module docstring says why. """ # Before the whitespace collapse below, which writes the field on one line and # leaves no line for _RULE to match. without_rules = _RULE.sub("", markdown) - folded = _MATHS.sub(_maths, without_rules.translate(_QUOTES)) + # Before the maths below, which drops the whitespace an entity has become. + spaces = _ENTITY.sub(" ", without_rules) + folded = _SPACED_MATHS.sub(_maths, spaces.translate(_QUOTES)) named = _IMAGE.sub(lambda reference: f"![]({Path(reference[1]).name})", folded) return " ".join(named.split()) diff --git a/in2lambda/main.py b/in2lambda/main.py index 95b2bd3..ed1a963 100644 --- a/in2lambda/main.py +++ b/in2lambda/main.py @@ -641,12 +641,13 @@ def compare(built_zip: str, export_dir: str, known_path: Optional[str]) -> None: printed naming the question, the part and the field. The differences in wording that are not differences in what a question says are taken off both sides first: a run of whitespace is compared as one space, a line of hyphens is dropped, a curly quote is - compared as a straight quote, the notation inside maths is folded, an image is - compared by the file's name, and a lone empty part is dropped. in2lambda.compare says - which notation and why. --known names a file of the differences the two sets are - known to have, one per line as this command prints it, with a ticket written after - " # ". in2lambda compare - exits 1 where the differences found are not the differences --known names. + compared as a straight quote, an HTML entity for a space is compared as a space, the + notation inside maths and the whitespace beside it are dropped, an image is compared + by the file's name, and a lone empty part is dropped. in2lambda.compare says which + notation and why. --known names a file of the differences the two sets are known to + have, one per line as this command prints it, with a ticket written after " # ". + in2lambda compare exits 1 where the differences found are not the differences --known + names. """ with _message_not_traceback(): found = in2lambda.compare.differences( diff --git a/tests/fixtures/against_convert/README.md b/tests/fixtures/against_convert/README.md index c525818..40bebe5 100644 --- a/tests/fixtures/against_convert/README.md +++ b/tests/fixtures/against_convert/README.md @@ -38,6 +38,11 @@ off both sides before comparing: - **Quotes.** Pandoc's LaTeX reader writes `’` where its `commonmark_x` writer writes `'`, so `aren’t` and `aren't` are the same wording. Each curly quote is compared as the straight quote. +- **HTML entities for a space.** Lambda Feedback writes ` ` and ` ` where a + document writes a space. Each entity is compared as a space. +- **Whitespace beside maths.** An export writes `equation of $y'+y=0$:` where convert + writes `equation of$y'+y=0$:`. The whitespace before and after each maths span is + dropped. - **Maths notation.** Inside every `$ ... $` and `$$ ... $$`, `\left` and `\right` are removed, `~`, `\,` and `\space` are compared as a space, and every run of whitespace is dropped, except that a run between a control word and a following letter is diff --git a/tests/test_compare.py b/tests/test_compare.py index d220747..d3e1953 100644 --- a/tests/test_compare.py +++ b/tests/test_compare.py @@ -53,6 +53,28 @@ def test_a_curly_quote_is_a_straight_quote() -> None: assert differences(_set("The “load”."), _set('The "load".')) == [] +def test_an_html_entity_for_a_space_is_a_space() -> None: + """Lambda Feedback writes ` ` and ` ` where a document writes a space.""" + assert ( + differences( + _set(r"$16y''-\pi^2y=0$ (Use $A$"), + _set(r"$16y''-\pi^2y=0$ (Use $A$"), + ) + == [] + ) + assert differences(_set("The load."), _set("The load.")) == [] + + +def test_whitespace_beside_maths_is_dropped() -> None: + """A space touching a `$` on the outside renders as no space.""" + assert ( + differences( + _set("equation of $y''+y'-6y=0$:"), _set("equation of$y''+y'-6y=0$:") + ) + == [] + ) + + def test_whitespace_inside_maths_is_dropped() -> None: """LaTeX renders `$z=2+3 i$` and `$z=2+3i$` the same.""" assert differences(_set("$z = 2+3 i$"), _set("$z=2+3i$")) == [] From 343626d17bc569f73dc92297d24b123e7eb3b437 Mon Sep 17 00:00:00 2001 From: "Peter B. Johnson" Date: Wed, 23 Sep 2026 23:19:45 +0100 Subject: [PATCH 3/6] implement: Fold notation in in2lambda's comparison (t59) --- CHANGELOG.md | 2 +- in2lambda/compare.py | 33 +++++------------------- in2lambda/main.py | 9 +++---- tests/fixtures/against_convert/README.md | 5 ---- tests/test_compare.py | 22 ---------------- 5 files changed, 12 insertions(+), 59 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index f5e6cbc..013e2b2 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -25,5 +25,5 @@ - `in2lambda convert FILE PartsOneSol` now exports the worked solution a document writes in a `solution` environment. Pandoc writes that environment as a Div whose classes hold `solution`, and the filter recognised only a Div whose first block reads `Solution`, so a document using the environment exported every question with an empty worked solution. A Div whose first block reads `Solution` is still recognised. - Importing `in2lambda.katex_convert` no longer writes a file called `log` into the working directory. That module reports what it changed in an expression to the `in2lambda.katex_convert` logger, which is silent unless the application configures logging. - `in2lambda convert` now reads a .docx that holds an image. in2lambda looks in the document for the directories a `\graphicspath` names, and read the document as UTF-8 text to find them. A .docx is a zip file, so converting a Word document holding a figure raised `UnicodeDecodeError`. in2lambda now reads a document that is not UTF-8 text as naming no directory, which is what a .docx names. -- `in2lambda compare BUILT_ZIP EXPORT_DIR` compares two Lambda Feedback sets, each given as a folder or a zip, so that a set in2lambda wrote can be checked against the export it should reproduce. Each question's main text is compared, and each part's text and worked solution, and every difference is printed naming the question, the part and the field. The differences in wording that are not differences in what a question says are taken off both sides first: a run of whitespace is compared as one space, a line holding nothing but hyphens is dropped, each curly quote is compared as the straight quote, ` ` and ` ` are compared as a space, an image is compared by the file's name, and a part holding neither text nor a worked solution is dropped where it is the question's only part. The whitespace before and after each maths span is dropped, so that `equation of $y'+y=0$:` and `equation of$y'+y=0$:` are the same wording. Inside every `$ ... $` and `$$ ... $$`, `\left` and `\right` are removed, `~`, `\,` and `\space` are compared as a space, and every run of whitespace is dropped, except that a run between a control word and a following letter is compared as one space, so that `$z=2+3 i$` and `$z=2+3i$` are the same expression and `$\alpha x$` and `$\alphax$` are two. `--known FILE` names the differences the two sets are known to have, one line per difference with the ticket that would close it written after ` # `, and `in2lambda compare` exits 1 where the differences found are not the differences that file names. `in2lambda.compare.differences` and `in2lambda.compare.known` are the two functions behind the command. +- `in2lambda compare BUILT_ZIP EXPORT_DIR` compares two Lambda Feedback sets, each given as a folder or a zip, so that a set in2lambda wrote can be checked against the export it should reproduce. Each question's main text is compared, and each part's text and worked solution, and every difference is printed naming the question, the part and the field. The differences in wording that are not differences in what a question says are taken off both sides first: a run of whitespace is compared as one space, a line holding nothing but hyphens is dropped, each curly quote is compared as the straight quote, an image is compared by the file's name, and a part holding neither text nor a worked solution is dropped where it is the question's only part. Inside every `$ ... $` and `$$ ... $$`, `\left` and `\right` are removed, `~`, `\,` and `\space` are compared as a space, and every run of whitespace is dropped, except that a run between a control word and a following letter is compared as one space, so that `$z=2+3 i$` and `$z=2+3i$` are the same expression and `$\alpha x$` and `$\alphax$` are two. `--known FILE` names the differences the two sets are known to have, one line per difference with the ticket that would close it written after ` # `, and `in2lambda compare` exits 1 where the differences found are not the differences that file names. `in2lambda.compare.differences` and `in2lambda.compare.known` are the two functions behind the command. - The rest of the Python API is unchanged: `in2lambda.main.runner` and everything under `in2lambda.api` take the same arguments and return the same objects. diff --git a/in2lambda/compare.py b/in2lambda/compare.py index b8b8862..5a162a5 100644 --- a/in2lambda/compare.py +++ b/in2lambda/compare.py @@ -18,12 +18,6 @@ - **Quotes.** ``‘`` and ``’`` are compared as ``'``, and ``“`` and ``”`` as ``"``, because pandoc's LaTeX reader writes the curly quote where its commonmark_x writer writes the straight one. -- **HTML entities for a space.** `` `` and `` `` are compared as a space, - because an export writes ``$y=0$ (Use $A$`` where `in2lambda convert` - writes ``$y=0$ (Use $A$``. -- **Whitespace beside maths.** The whitespace before and after each ``$ ... $`` and - ``$$ ... $$`` is dropped, because an export writes ``equation of $y'+y=0$:`` where - `in2lambda convert` writes ``equation of$y'+y=0$:``. - **Maths notation.** Inside every ``$ ... $`` and ``$$ ... $$``, ``\left`` and ``\right`` are removed, ``~``, ``\,`` and ``\space`` are compared as a space, and every run of whitespace is dropped, except that a run between a control word and a @@ -63,14 +57,6 @@ _QUOTES = str.maketrans({"‘": "'", "’": "'", "“": '"', "”": '"'}) """Each curly quote and the straight quote it is compared as.""" -_ENTITY = re.compile(r" | ") -"""The two HTML entities Lambda Feedback writes a space as.""" - -_SPACED_MATHS = re.compile(rf"\s*(?:{_MATHS.pattern})\s*", re.DOTALL) -"""One maths span with the whitespace touching it. Wrapping `in2lambda.validation._MATHS` -in a group that captures nothing keeps the display maths it found in the first group and -the inline maths in the second.""" - _SIZE = re.compile(r"\\(?:left|right)(?![a-zA-Z])") r"""``\left`` and ``\right``, which size a delimiter without changing which it is.""" @@ -94,11 +80,9 @@ def _spacing(whitespace: re.Match[str]) -> str: def _maths(expression: re.Match[str]) -> str: """One ``$ ... $`` or ``$$ ... $$`` with the notation that is not the maths folded. - The whitespace `_SPACED_MATHS` matched around the span is dropped with it. - Args: - expression: A match of `_SPACED_MATHS`, holding the display maths it found in - its first group and the inline maths in its second. + expression: A match of `in2lambda.validation._MATHS`, holding the display maths + it found in its first group and the inline maths in its second. """ display = expression[1] is not None tex = expression[1] if display else expression[2] @@ -110,18 +94,15 @@ def _maths(expression: re.Match[str]) -> str: def _text(markdown: str) -> str: """A field with the differences in wording that are not differences taken off. - A separator line is dropped, each HTML entity for a space becomes a space, each - curly quote becomes a straight quote, the maths notation inside every ``$ ... $`` - and ``$$ ... $$`` is folded with the whitespace touching the span, every image - reference is written as the file's name alone, and every run of whitespace becomes - one space. The module docstring says why. + A separator line is dropped, each curly quote becomes a straight quote, the maths + notation inside every ``$ ... $`` and ``$$ ... $$`` is folded, every image reference + is written as the file's name alone, and every run of whitespace becomes one space. + The module docstring says why. """ # Before the whitespace collapse below, which writes the field on one line and # leaves no line for _RULE to match. without_rules = _RULE.sub("", markdown) - # Before the maths below, which drops the whitespace an entity has become. - spaces = _ENTITY.sub(" ", without_rules) - folded = _SPACED_MATHS.sub(_maths, spaces.translate(_QUOTES)) + folded = _MATHS.sub(_maths, without_rules.translate(_QUOTES)) named = _IMAGE.sub(lambda reference: f"![]({Path(reference[1]).name})", folded) return " ".join(named.split()) diff --git a/in2lambda/main.py b/in2lambda/main.py index ed1a963..67b6468 100644 --- a/in2lambda/main.py +++ b/in2lambda/main.py @@ -641,11 +641,10 @@ def compare(built_zip: str, export_dir: str, known_path: Optional[str]) -> None: printed naming the question, the part and the field. The differences in wording that are not differences in what a question says are taken off both sides first: a run of whitespace is compared as one space, a line of hyphens is dropped, a curly quote is - compared as a straight quote, an HTML entity for a space is compared as a space, the - notation inside maths and the whitespace beside it are dropped, an image is compared - by the file's name, and a lone empty part is dropped. in2lambda.compare says which - notation and why. --known names a file of the differences the two sets are known to - have, one per line as this command prints it, with a ticket written after " # ". + compared as a straight quote, the notation inside maths is dropped, an image is + compared by the file's name, and a lone empty part is dropped. in2lambda.compare says + which notation and why. --known names a file of the differences the two sets are known + to have, one per line as this command prints it, with a ticket written after " # ". in2lambda compare exits 1 where the differences found are not the differences --known names. """ diff --git a/tests/fixtures/against_convert/README.md b/tests/fixtures/against_convert/README.md index 40bebe5..c525818 100644 --- a/tests/fixtures/against_convert/README.md +++ b/tests/fixtures/against_convert/README.md @@ -38,11 +38,6 @@ off both sides before comparing: - **Quotes.** Pandoc's LaTeX reader writes `’` where its `commonmark_x` writer writes `'`, so `aren’t` and `aren't` are the same wording. Each curly quote is compared as the straight quote. -- **HTML entities for a space.** Lambda Feedback writes ` ` and ` ` where a - document writes a space. Each entity is compared as a space. -- **Whitespace beside maths.** An export writes `equation of $y'+y=0$:` where convert - writes `equation of$y'+y=0$:`. The whitespace before and after each maths span is - dropped. - **Maths notation.** Inside every `$ ... $` and `$$ ... $$`, `\left` and `\right` are removed, `~`, `\,` and `\space` are compared as a space, and every run of whitespace is dropped, except that a run between a control word and a following letter is diff --git a/tests/test_compare.py b/tests/test_compare.py index d3e1953..d220747 100644 --- a/tests/test_compare.py +++ b/tests/test_compare.py @@ -53,28 +53,6 @@ def test_a_curly_quote_is_a_straight_quote() -> None: assert differences(_set("The “load”."), _set('The "load".')) == [] -def test_an_html_entity_for_a_space_is_a_space() -> None: - """Lambda Feedback writes ` ` and ` ` where a document writes a space.""" - assert ( - differences( - _set(r"$16y''-\pi^2y=0$ (Use $A$"), - _set(r"$16y''-\pi^2y=0$ (Use $A$"), - ) - == [] - ) - assert differences(_set("The load."), _set("The load.")) == [] - - -def test_whitespace_beside_maths_is_dropped() -> None: - """A space touching a `$` on the outside renders as no space.""" - assert ( - differences( - _set("equation of $y''+y'-6y=0$:"), _set("equation of$y''+y'-6y=0$:") - ) - == [] - ) - - def test_whitespace_inside_maths_is_dropped() -> None: """LaTeX renders `$z=2+3 i$` and `$z=2+3i$` the same.""" assert differences(_set("$z = 2+3 i$"), _set("$z=2+3i$")) == [] From 897fc0559ac6c3a705a9c99c51b540109bb4827b Mon Sep 17 00:00:00 2001 From: "Peter B. Johnson" Date: Wed, 23 Sep 2026 23:35:08 +0100 Subject: [PATCH 4/6] implement: Fold notation in in2lambda's comparison (t59) --- CHANGELOG.md | 2 +- in2lambda/compare.py | 33 +++++++++++++++++++----- in2lambda/main.py | 9 ++++--- tests/fixtures/against_convert/README.md | 5 ++++ tests/test_compare.py | 26 +++++++++++++++++++ 5 files changed, 63 insertions(+), 12 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 013e2b2..f5e6cbc 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -25,5 +25,5 @@ - `in2lambda convert FILE PartsOneSol` now exports the worked solution a document writes in a `solution` environment. Pandoc writes that environment as a Div whose classes hold `solution`, and the filter recognised only a Div whose first block reads `Solution`, so a document using the environment exported every question with an empty worked solution. A Div whose first block reads `Solution` is still recognised. - Importing `in2lambda.katex_convert` no longer writes a file called `log` into the working directory. That module reports what it changed in an expression to the `in2lambda.katex_convert` logger, which is silent unless the application configures logging. - `in2lambda convert` now reads a .docx that holds an image. in2lambda looks in the document for the directories a `\graphicspath` names, and read the document as UTF-8 text to find them. A .docx is a zip file, so converting a Word document holding a figure raised `UnicodeDecodeError`. in2lambda now reads a document that is not UTF-8 text as naming no directory, which is what a .docx names. -- `in2lambda compare BUILT_ZIP EXPORT_DIR` compares two Lambda Feedback sets, each given as a folder or a zip, so that a set in2lambda wrote can be checked against the export it should reproduce. Each question's main text is compared, and each part's text and worked solution, and every difference is printed naming the question, the part and the field. The differences in wording that are not differences in what a question says are taken off both sides first: a run of whitespace is compared as one space, a line holding nothing but hyphens is dropped, each curly quote is compared as the straight quote, an image is compared by the file's name, and a part holding neither text nor a worked solution is dropped where it is the question's only part. Inside every `$ ... $` and `$$ ... $$`, `\left` and `\right` are removed, `~`, `\,` and `\space` are compared as a space, and every run of whitespace is dropped, except that a run between a control word and a following letter is compared as one space, so that `$z=2+3 i$` and `$z=2+3i$` are the same expression and `$\alpha x$` and `$\alphax$` are two. `--known FILE` names the differences the two sets are known to have, one line per difference with the ticket that would close it written after ` # `, and `in2lambda compare` exits 1 where the differences found are not the differences that file names. `in2lambda.compare.differences` and `in2lambda.compare.known` are the two functions behind the command. +- `in2lambda compare BUILT_ZIP EXPORT_DIR` compares two Lambda Feedback sets, each given as a folder or a zip, so that a set in2lambda wrote can be checked against the export it should reproduce. Each question's main text is compared, and each part's text and worked solution, and every difference is printed naming the question, the part and the field. The differences in wording that are not differences in what a question says are taken off both sides first: a run of whitespace is compared as one space, a line holding nothing but hyphens is dropped, each curly quote is compared as the straight quote, ` ` and ` ` are compared as a space, an image is compared by the file's name, and a part holding neither text nor a worked solution is dropped where it is the question's only part. The whitespace before and after each maths span is dropped, so that `equation of $y'+y=0$:` and `equation of$y'+y=0$:` are the same wording. Inside every `$ ... $` and `$$ ... $$`, `\left` and `\right` are removed, `~`, `\,` and `\space` are compared as a space, and every run of whitespace is dropped, except that a run between a control word and a following letter is compared as one space, so that `$z=2+3 i$` and `$z=2+3i$` are the same expression and `$\alpha x$` and `$\alphax$` are two. `--known FILE` names the differences the two sets are known to have, one line per difference with the ticket that would close it written after ` # `, and `in2lambda compare` exits 1 where the differences found are not the differences that file names. `in2lambda.compare.differences` and `in2lambda.compare.known` are the two functions behind the command. - The rest of the Python API is unchanged: `in2lambda.main.runner` and everything under `in2lambda.api` take the same arguments and return the same objects. diff --git a/in2lambda/compare.py b/in2lambda/compare.py index 5a162a5..b8b8862 100644 --- a/in2lambda/compare.py +++ b/in2lambda/compare.py @@ -18,6 +18,12 @@ - **Quotes.** ``‘`` and ``’`` are compared as ``'``, and ``“`` and ``”`` as ``"``, because pandoc's LaTeX reader writes the curly quote where its commonmark_x writer writes the straight one. +- **HTML entities for a space.** `` `` and `` `` are compared as a space, + because an export writes ``$y=0$ (Use $A$`` where `in2lambda convert` + writes ``$y=0$ (Use $A$``. +- **Whitespace beside maths.** The whitespace before and after each ``$ ... $`` and + ``$$ ... $$`` is dropped, because an export writes ``equation of $y'+y=0$:`` where + `in2lambda convert` writes ``equation of$y'+y=0$:``. - **Maths notation.** Inside every ``$ ... $`` and ``$$ ... $$``, ``\left`` and ``\right`` are removed, ``~``, ``\,`` and ``\space`` are compared as a space, and every run of whitespace is dropped, except that a run between a control word and a @@ -57,6 +63,14 @@ _QUOTES = str.maketrans({"‘": "'", "’": "'", "“": '"', "”": '"'}) """Each curly quote and the straight quote it is compared as.""" +_ENTITY = re.compile(r" | ") +"""The two HTML entities Lambda Feedback writes a space as.""" + +_SPACED_MATHS = re.compile(rf"\s*(?:{_MATHS.pattern})\s*", re.DOTALL) +"""One maths span with the whitespace touching it. Wrapping `in2lambda.validation._MATHS` +in a group that captures nothing keeps the display maths it found in the first group and +the inline maths in the second.""" + _SIZE = re.compile(r"\\(?:left|right)(?![a-zA-Z])") r"""``\left`` and ``\right``, which size a delimiter without changing which it is.""" @@ -80,9 +94,11 @@ def _spacing(whitespace: re.Match[str]) -> str: def _maths(expression: re.Match[str]) -> str: """One ``$ ... $`` or ``$$ ... $$`` with the notation that is not the maths folded. + The whitespace `_SPACED_MATHS` matched around the span is dropped with it. + Args: - expression: A match of `in2lambda.validation._MATHS`, holding the display maths - it found in its first group and the inline maths in its second. + expression: A match of `_SPACED_MATHS`, holding the display maths it found in + its first group and the inline maths in its second. """ display = expression[1] is not None tex = expression[1] if display else expression[2] @@ -94,15 +110,18 @@ def _maths(expression: re.Match[str]) -> str: def _text(markdown: str) -> str: """A field with the differences in wording that are not differences taken off. - A separator line is dropped, each curly quote becomes a straight quote, the maths - notation inside every ``$ ... $`` and ``$$ ... $$`` is folded, every image reference - is written as the file's name alone, and every run of whitespace becomes one space. - The module docstring says why. + A separator line is dropped, each HTML entity for a space becomes a space, each + curly quote becomes a straight quote, the maths notation inside every ``$ ... $`` + and ``$$ ... $$`` is folded with the whitespace touching the span, every image + reference is written as the file's name alone, and every run of whitespace becomes + one space. The module docstring says why. """ # Before the whitespace collapse below, which writes the field on one line and # leaves no line for _RULE to match. without_rules = _RULE.sub("", markdown) - folded = _MATHS.sub(_maths, without_rules.translate(_QUOTES)) + # Before the maths below, which drops the whitespace an entity has become. + spaces = _ENTITY.sub(" ", without_rules) + folded = _SPACED_MATHS.sub(_maths, spaces.translate(_QUOTES)) named = _IMAGE.sub(lambda reference: f"![]({Path(reference[1]).name})", folded) return " ".join(named.split()) diff --git a/in2lambda/main.py b/in2lambda/main.py index 67b6468..ed1a963 100644 --- a/in2lambda/main.py +++ b/in2lambda/main.py @@ -641,10 +641,11 @@ def compare(built_zip: str, export_dir: str, known_path: Optional[str]) -> None: printed naming the question, the part and the field. The differences in wording that are not differences in what a question says are taken off both sides first: a run of whitespace is compared as one space, a line of hyphens is dropped, a curly quote is - compared as a straight quote, the notation inside maths is dropped, an image is - compared by the file's name, and a lone empty part is dropped. in2lambda.compare says - which notation and why. --known names a file of the differences the two sets are known - to have, one per line as this command prints it, with a ticket written after " # ". + compared as a straight quote, an HTML entity for a space is compared as a space, the + notation inside maths and the whitespace beside it are dropped, an image is compared + by the file's name, and a lone empty part is dropped. in2lambda.compare says which + notation and why. --known names a file of the differences the two sets are known to + have, one per line as this command prints it, with a ticket written after " # ". in2lambda compare exits 1 where the differences found are not the differences --known names. """ diff --git a/tests/fixtures/against_convert/README.md b/tests/fixtures/against_convert/README.md index c525818..40bebe5 100644 --- a/tests/fixtures/against_convert/README.md +++ b/tests/fixtures/against_convert/README.md @@ -38,6 +38,11 @@ off both sides before comparing: - **Quotes.** Pandoc's LaTeX reader writes `’` where its `commonmark_x` writer writes `'`, so `aren’t` and `aren't` are the same wording. Each curly quote is compared as the straight quote. +- **HTML entities for a space.** Lambda Feedback writes ` ` and ` ` where a + document writes a space. Each entity is compared as a space. +- **Whitespace beside maths.** An export writes `equation of $y'+y=0$:` where convert + writes `equation of$y'+y=0$:`. The whitespace before and after each maths span is + dropped. - **Maths notation.** Inside every `$ ... $` and `$$ ... $$`, `\left` and `\right` are removed, `~`, `\,` and `\space` are compared as a space, and every run of whitespace is dropped, except that a run between a control word and a following letter is diff --git a/tests/test_compare.py b/tests/test_compare.py index d220747..d39281a 100644 --- a/tests/test_compare.py +++ b/tests/test_compare.py @@ -53,6 +53,32 @@ def test_a_curly_quote_is_a_straight_quote() -> None: assert differences(_set("The “load”."), _set('The "load".')) == [] +def test_an_html_entity_for_a_space_is_a_space() -> None: + """Lambda Feedback writes ` ` and ` ` where a document writes a space.""" + assert ( + differences( + _set( + r"$16y''-\pi^2y=0$ " + "(Use $A$ and $B$ for your constants.)" + ), + _set(r"$16y''-\pi^2y=0$ (Use $A$ and $B$ for your constants.)"), + ) + == [] + ) + assert differences(_set("The load."), _set("The load.")) == [] + + +def test_whitespace_beside_maths_is_dropped() -> None: + """A space touching a `$` on the outside renders as no space.""" + assert ( + differences( + _set("equation of $y''+y'-6y=0$: then"), + _set("equation of$y''+y'-6y=0$: then"), + ) + == [] + ) + + def test_whitespace_inside_maths_is_dropped() -> None: """LaTeX renders `$z=2+3 i$` and `$z=2+3i$` the same.""" assert differences(_set("$z = 2+3 i$"), _set("$z=2+3i$")) == [] From ae0f1ed8365153968bdb11f2d4bb34d518d21c8e Mon Sep 17 00:00:00 2001 From: "Peter B. Johnson" Date: Wed, 23 Sep 2026 23:49:00 +0100 Subject: [PATCH 5/6] implement: Fold notation in in2lambda's comparison (t59) --- CHANGELOG.md | 2 +- in2lambda/compare.py | 33 +++++------------------- in2lambda/main.py | 7 +++-- tests/fixtures/against_convert/README.md | 5 ---- tests/test_compare.py | 26 ------------------- 5 files changed, 11 insertions(+), 62 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index f5e6cbc..013e2b2 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -25,5 +25,5 @@ - `in2lambda convert FILE PartsOneSol` now exports the worked solution a document writes in a `solution` environment. Pandoc writes that environment as a Div whose classes hold `solution`, and the filter recognised only a Div whose first block reads `Solution`, so a document using the environment exported every question with an empty worked solution. A Div whose first block reads `Solution` is still recognised. - Importing `in2lambda.katex_convert` no longer writes a file called `log` into the working directory. That module reports what it changed in an expression to the `in2lambda.katex_convert` logger, which is silent unless the application configures logging. - `in2lambda convert` now reads a .docx that holds an image. in2lambda looks in the document for the directories a `\graphicspath` names, and read the document as UTF-8 text to find them. A .docx is a zip file, so converting a Word document holding a figure raised `UnicodeDecodeError`. in2lambda now reads a document that is not UTF-8 text as naming no directory, which is what a .docx names. -- `in2lambda compare BUILT_ZIP EXPORT_DIR` compares two Lambda Feedback sets, each given as a folder or a zip, so that a set in2lambda wrote can be checked against the export it should reproduce. Each question's main text is compared, and each part's text and worked solution, and every difference is printed naming the question, the part and the field. The differences in wording that are not differences in what a question says are taken off both sides first: a run of whitespace is compared as one space, a line holding nothing but hyphens is dropped, each curly quote is compared as the straight quote, ` ` and ` ` are compared as a space, an image is compared by the file's name, and a part holding neither text nor a worked solution is dropped where it is the question's only part. The whitespace before and after each maths span is dropped, so that `equation of $y'+y=0$:` and `equation of$y'+y=0$:` are the same wording. Inside every `$ ... $` and `$$ ... $$`, `\left` and `\right` are removed, `~`, `\,` and `\space` are compared as a space, and every run of whitespace is dropped, except that a run between a control word and a following letter is compared as one space, so that `$z=2+3 i$` and `$z=2+3i$` are the same expression and `$\alpha x$` and `$\alphax$` are two. `--known FILE` names the differences the two sets are known to have, one line per difference with the ticket that would close it written after ` # `, and `in2lambda compare` exits 1 where the differences found are not the differences that file names. `in2lambda.compare.differences` and `in2lambda.compare.known` are the two functions behind the command. +- `in2lambda compare BUILT_ZIP EXPORT_DIR` compares two Lambda Feedback sets, each given as a folder or a zip, so that a set in2lambda wrote can be checked against the export it should reproduce. Each question's main text is compared, and each part's text and worked solution, and every difference is printed naming the question, the part and the field. The differences in wording that are not differences in what a question says are taken off both sides first: a run of whitespace is compared as one space, a line holding nothing but hyphens is dropped, each curly quote is compared as the straight quote, an image is compared by the file's name, and a part holding neither text nor a worked solution is dropped where it is the question's only part. Inside every `$ ... $` and `$$ ... $$`, `\left` and `\right` are removed, `~`, `\,` and `\space` are compared as a space, and every run of whitespace is dropped, except that a run between a control word and a following letter is compared as one space, so that `$z=2+3 i$` and `$z=2+3i$` are the same expression and `$\alpha x$` and `$\alphax$` are two. `--known FILE` names the differences the two sets are known to have, one line per difference with the ticket that would close it written after ` # `, and `in2lambda compare` exits 1 where the differences found are not the differences that file names. `in2lambda.compare.differences` and `in2lambda.compare.known` are the two functions behind the command. - The rest of the Python API is unchanged: `in2lambda.main.runner` and everything under `in2lambda.api` take the same arguments and return the same objects. diff --git a/in2lambda/compare.py b/in2lambda/compare.py index b8b8862..5a162a5 100644 --- a/in2lambda/compare.py +++ b/in2lambda/compare.py @@ -18,12 +18,6 @@ - **Quotes.** ``‘`` and ``’`` are compared as ``'``, and ``“`` and ``”`` as ``"``, because pandoc's LaTeX reader writes the curly quote where its commonmark_x writer writes the straight one. -- **HTML entities for a space.** `` `` and `` `` are compared as a space, - because an export writes ``$y=0$ (Use $A$`` where `in2lambda convert` - writes ``$y=0$ (Use $A$``. -- **Whitespace beside maths.** The whitespace before and after each ``$ ... $`` and - ``$$ ... $$`` is dropped, because an export writes ``equation of $y'+y=0$:`` where - `in2lambda convert` writes ``equation of$y'+y=0$:``. - **Maths notation.** Inside every ``$ ... $`` and ``$$ ... $$``, ``\left`` and ``\right`` are removed, ``~``, ``\,`` and ``\space`` are compared as a space, and every run of whitespace is dropped, except that a run between a control word and a @@ -63,14 +57,6 @@ _QUOTES = str.maketrans({"‘": "'", "’": "'", "“": '"', "”": '"'}) """Each curly quote and the straight quote it is compared as.""" -_ENTITY = re.compile(r" | ") -"""The two HTML entities Lambda Feedback writes a space as.""" - -_SPACED_MATHS = re.compile(rf"\s*(?:{_MATHS.pattern})\s*", re.DOTALL) -"""One maths span with the whitespace touching it. Wrapping `in2lambda.validation._MATHS` -in a group that captures nothing keeps the display maths it found in the first group and -the inline maths in the second.""" - _SIZE = re.compile(r"\\(?:left|right)(?![a-zA-Z])") r"""``\left`` and ``\right``, which size a delimiter without changing which it is.""" @@ -94,11 +80,9 @@ def _spacing(whitespace: re.Match[str]) -> str: def _maths(expression: re.Match[str]) -> str: """One ``$ ... $`` or ``$$ ... $$`` with the notation that is not the maths folded. - The whitespace `_SPACED_MATHS` matched around the span is dropped with it. - Args: - expression: A match of `_SPACED_MATHS`, holding the display maths it found in - its first group and the inline maths in its second. + expression: A match of `in2lambda.validation._MATHS`, holding the display maths + it found in its first group and the inline maths in its second. """ display = expression[1] is not None tex = expression[1] if display else expression[2] @@ -110,18 +94,15 @@ def _maths(expression: re.Match[str]) -> str: def _text(markdown: str) -> str: """A field with the differences in wording that are not differences taken off. - A separator line is dropped, each HTML entity for a space becomes a space, each - curly quote becomes a straight quote, the maths notation inside every ``$ ... $`` - and ``$$ ... $$`` is folded with the whitespace touching the span, every image - reference is written as the file's name alone, and every run of whitespace becomes - one space. The module docstring says why. + A separator line is dropped, each curly quote becomes a straight quote, the maths + notation inside every ``$ ... $`` and ``$$ ... $$`` is folded, every image reference + is written as the file's name alone, and every run of whitespace becomes one space. + The module docstring says why. """ # Before the whitespace collapse below, which writes the field on one line and # leaves no line for _RULE to match. without_rules = _RULE.sub("", markdown) - # Before the maths below, which drops the whitespace an entity has become. - spaces = _ENTITY.sub(" ", without_rules) - folded = _SPACED_MATHS.sub(_maths, spaces.translate(_QUOTES)) + folded = _MATHS.sub(_maths, without_rules.translate(_QUOTES)) named = _IMAGE.sub(lambda reference: f"![]({Path(reference[1]).name})", folded) return " ".join(named.split()) diff --git a/in2lambda/main.py b/in2lambda/main.py index ed1a963..1227389 100644 --- a/in2lambda/main.py +++ b/in2lambda/main.py @@ -641,10 +641,9 @@ def compare(built_zip: str, export_dir: str, known_path: Optional[str]) -> None: printed naming the question, the part and the field. The differences in wording that are not differences in what a question says are taken off both sides first: a run of whitespace is compared as one space, a line of hyphens is dropped, a curly quote is - compared as a straight quote, an HTML entity for a space is compared as a space, the - notation inside maths and the whitespace beside it are dropped, an image is compared - by the file's name, and a lone empty part is dropped. in2lambda.compare says which - notation and why. --known names a file of the differences the two sets are known to + compared as a straight quote, the notation inside maths is folded, an image is + compared by the file's name, and a lone empty part is dropped. in2lambda.compare says + which notation and why. --known names a file of the differences the two sets are known to have, one per line as this command prints it, with a ticket written after " # ". in2lambda compare exits 1 where the differences found are not the differences --known names. diff --git a/tests/fixtures/against_convert/README.md b/tests/fixtures/against_convert/README.md index 40bebe5..c525818 100644 --- a/tests/fixtures/against_convert/README.md +++ b/tests/fixtures/against_convert/README.md @@ -38,11 +38,6 @@ off both sides before comparing: - **Quotes.** Pandoc's LaTeX reader writes `’` where its `commonmark_x` writer writes `'`, so `aren’t` and `aren't` are the same wording. Each curly quote is compared as the straight quote. -- **HTML entities for a space.** Lambda Feedback writes ` ` and ` ` where a - document writes a space. Each entity is compared as a space. -- **Whitespace beside maths.** An export writes `equation of $y'+y=0$:` where convert - writes `equation of$y'+y=0$:`. The whitespace before and after each maths span is - dropped. - **Maths notation.** Inside every `$ ... $` and `$$ ... $$`, `\left` and `\right` are removed, `~`, `\,` and `\space` are compared as a space, and every run of whitespace is dropped, except that a run between a control word and a following letter is diff --git a/tests/test_compare.py b/tests/test_compare.py index d39281a..d220747 100644 --- a/tests/test_compare.py +++ b/tests/test_compare.py @@ -53,32 +53,6 @@ def test_a_curly_quote_is_a_straight_quote() -> None: assert differences(_set("The “load”."), _set('The "load".')) == [] -def test_an_html_entity_for_a_space_is_a_space() -> None: - """Lambda Feedback writes ` ` and ` ` where a document writes a space.""" - assert ( - differences( - _set( - r"$16y''-\pi^2y=0$ " - "(Use $A$ and $B$ for your constants.)" - ), - _set(r"$16y''-\pi^2y=0$ (Use $A$ and $B$ for your constants.)"), - ) - == [] - ) - assert differences(_set("The load."), _set("The load.")) == [] - - -def test_whitespace_beside_maths_is_dropped() -> None: - """A space touching a `$` on the outside renders as no space.""" - assert ( - differences( - _set("equation of $y''+y'-6y=0$: then"), - _set("equation of$y''+y'-6y=0$: then"), - ) - == [] - ) - - def test_whitespace_inside_maths_is_dropped() -> None: """LaTeX renders `$z=2+3 i$` and `$z=2+3i$` the same.""" assert differences(_set("$z = 2+3 i$"), _set("$z=2+3i$")) == [] From 426466039f1e51cfe658f8749ab40527bc3d820e Mon Sep 17 00:00:00 2001 From: "Peter B. Johnson" Date: Wed, 23 Sep 2026 23:59:38 +0100 Subject: [PATCH 6/6] Fold HTML space entities, whitespace beside maths delimiters and asterisk separators in the comparison CW2 of the EART40013 targets compares with 1 difference where it compared with 17. Co-Authored-By: Claude Fable 5.1 --- in2lambda/compare.py | 11 ++++++++--- tests/test_compare.py | 18 ++++++++++++++++++ 2 files changed, 26 insertions(+), 3 deletions(-) diff --git a/in2lambda/compare.py b/in2lambda/compare.py index 5a162a5..e986db9 100644 --- a/in2lambda/compare.py +++ b/in2lambda/compare.py @@ -51,8 +51,9 @@ _TICKET = " # " """What a line of a differs.txt names the ticket closing it after.""" -_RULE = re.compile(r"(?m)^[ \t]*-{3,}[ \t]*$") -"""A line holding nothing but hyphens, which is the separator Lambda Feedback writes.""" +_RULE = re.compile(r"(?m)^[ \t]*(?:-{3,}|\*{3,}|_{3,})[ \t]*$") +"""A line holding nothing but three or more hyphens, asterisks or underscores: a markdown +separator, which Lambda Feedback writes around a display maths as `---` or `***`.""" _QUOTES = str.maketrans({"‘": "'", "’": "'", "“": '"', "”": '"'}) """Each curly quote and the straight quote it is compared as.""" @@ -101,8 +102,12 @@ def _text(markdown: str) -> str: """ # Before the whitespace collapse below, which writes the field on one line and # leaves no line for _RULE to match. - without_rules = _RULE.sub("", markdown) + # The platform writes a space as the entity ` ` (and a hard one as ` `). + without_rules = _RULE.sub("", markdown.replace(" ", " ").replace(" ", " ")) folded = _MATHS.sub(_maths, without_rules.translate(_QUOTES)) + # A space touching a maths delimiter from outside renders the same either way: + # `of $y$` and `of$y$` are one expression, so the whitespace beside `$` is dropped. + folded = re.sub(r"\s*(\${1,2})\s*", r"\1", folded) named = _IMAGE.sub(lambda reference: f"![]({Path(reference[1]).name})", folded) return " ".join(named.split()) diff --git a/tests/test_compare.py b/tests/test_compare.py index d220747..78e18ec 100644 --- a/tests/test_compare.py +++ b/tests/test_compare.py @@ -209,3 +209,21 @@ def test_the_command_refuses_a_path_that_is_not_a_set( result.exception, (ValueError, zipfile.BadZipFile) ), result.output assert f"{path} is not a Lambda Feedback set" in result.output + + +def test_an_html_space_entity_is_a_space() -> None: + a = _set("$16y''-\\pi^2y=0$ (Use $A$ and $B$ for your constants.)") + b = _set("$16y''-\\pi^2y=0$ (Use $A$ and $B$ for your constants.)") + assert differences(a, b) == [] + + +def test_whitespace_touching_a_maths_delimiter_from_outside_is_ignored() -> None: + a = _set("equation of $y''+y'-6y=0$: then") + b = _set("equation of$y''+y'-6y=0$: then") + assert differences(a, b) == [] + + +def test_a_separator_of_asterisks_is_dropped_like_one_of_hyphens() -> None: + a = _set("Then:\n\n$$\nx=1\n$$\n\n***\n\nRecall the rule.") + b = _set("Then:\n\n$$\nx=1\n$$\n\nRecall the rule.") + assert differences(a, b) == []