diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..2749c66 --- /dev/null +++ b/.gitattributes @@ -0,0 +1,2 @@ +# The fixture is CRLF because that is the thing it is for; nothing may rewrite it. +tests/fixtures/sources/crlf/source.md -text diff --git a/CHANGELOG.md b/CHANGELOG.md index cc2ce88..f95b615 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -5,4 +5,5 @@ - Converting a document is now `in2lambda convert FILE FILTER`, with the same options as before (`-o/--out`, `-a/--answers`). Scripts and Docker invocations that run `in2lambda FILE FILTER` need the extra word. - `in2lambda FILE FILTER` exits with an error naming the command to run instead, rather than printing its usage and exiting successfully. - beartype is now `^0.22`. At 0.20.0 and below its import hook leaves `cli` a plain function rather than a group, so the new command line either fails to import or runs `convert` whatever the arguments; 0.20.1 is the first version that works. +- `in2lambda source add FILE` freezes a document: it converts .docx and .tex to markdown beside the file, and writes a `draft.json` holding the markdown's hash and every block in it with the lines it spans, so that another tool can quote the source by line range. `in2lambda source show` prints that markdown numbered with the block ids. Freezing a file that has changed since is refused unless `--start-over` says to discard the draft, and so is showing one, since its block ids would name lines they are not the ids of. Both need pandoc and the `convert` extra, as `convert` does. - The Python API is unchanged: `in2lambda.main.runner` and everything under `in2lambda.api` take the same arguments and return the same objects. diff --git a/in2lambda/main.py b/in2lambda/main.py index 52b1de0..9ed757c 100644 --- a/in2lambda/main.py +++ b/in2lambda/main.py @@ -7,32 +7,24 @@ # sys.path.insert(0, os.path.abspath(os.path.join(os.path.dirname(__file__), '..'))) import importlib -import importlib.util import shlex -import shutil -import subprocess from typing import Optional import rich_click as click import in2lambda.filters +import in2lambda.source from in2lambda.api.set import Set - -class ConversionToolsMissing(RuntimeError): - """Document conversion was asked for without pandoc or panflute installed.""" - - -def _require_conversion_tools() -> None: - missing = [] - if shutil.which("pandoc") is None: - missing.append("pandoc (see https://pandoc.org/installing.html)") - if importlib.util.find_spec("panflute") is None: - missing.append("panflute (pip install 'in2lambda[convert]')") - if missing: - raise ConversionToolsMissing( - f"Converting documents needs {' and '.join(missing)}." - ) +# Both were defined here before there was an in2lambda.source, and are in other +# people's scripts as in2lambda.main names. +from in2lambda.source import ( + ConversionToolsMissing, + SourceError, + _pandoc, + _require_conversion_tools, + file_type, +) def docx_to_md(docx_file: str) -> str: @@ -44,53 +36,7 @@ def docx_to_md(docx_file: str) -> str: Returns: the contents of the .docx file in markdown formatting """ - md_output = subprocess.check_output(["pandoc", docx_file, "-t", "markdown"]) - return md_output.decode("utf-8") - - -def file_type(file: str) -> str: - """Determines which pandoc file format to use for a given file. - - See https://github.com/jgm/pandoc/blob/bad922a69236e22b20d51c4ec0b90c5a6c038433/src/Text/Pandoc/Format.hs#L171 - (or any newer commit) for pandoc's supported file extensions. - - Args: - file: A file path with the file extension included. - - Returns: - An option in `pandoc --list-input-formats` that matches the given file type - - Examples: - >>> from in2lambda.main import file_type - >>> file_type("example.tex") - 'latex' - >>> file_type("/some/random/path/demo.md") - 'markdown' - >>> file_type("no_extension") - Traceback (most recent call last): - RuntimeError: Unsupported file extension: .no_extension - >>> file_type("demo.unknown_extension") - Traceback (most recent call last): - RuntimeError: Unsupported file extension: .unknown_extension - """ - match (extension := file.split(".")[-1].lower()): - case "tex" | "latex" | "ltx": - return "latex" - case ( - "md" - | "rmd" - | "markdown" - | "mdown" - | "mdwn" - | "mkd" - | "mkdn" - | "text" - | "txt" - ): - return "markdown" - case "docx": - return "docx" # Pandoc doesn't seem to support .doc, and panflute doesn't like .docx. - raise RuntimeError(f"Unsupported file extension: .{extension}") + return _pandoc(docx_file, "markdown").decode("utf-8") def runner( @@ -154,9 +100,7 @@ def runner( # If separate answer TeX file provided, parse that as well. if answer_file: - if file_type(answer_file) == "docx": - answer_text = docx_to_md(answer_file) answer_format = "markdown" else: @@ -251,5 +195,36 @@ def convert( raise click.ClickException(str(error)) from None +@cli.group("source") +def source_group() -> None: + """Freezes a source document, so its text can be quoted by line range.""" + + +@source_group.command("add") +@click.argument("file", type=click.Path(exists=True, dir_okay=False, resolve_path=True)) +@click.option( + "--start-over", + is_flag=True, + help="Freeze FILE again, discarding the draft already there.", +) +def source_add(file: str, start_over: bool) -> None: + """Converts FILE to markdown and records its blocks in draft.json beside it.""" + try: + draft = in2lambda.source.add(file, start_over) + except SourceError as error: + # Exit with what to do about it rather than a traceback. + raise click.ClickException(str(error)) from None + click.echo(f"Wrote {draft}") + + +@source_group.command("show") +def source_show() -> None: + """Prints the frozen markdown of the draft in this directory, numbered.""" + try: + click.echo(in2lambda.source.show()) + except SourceError as error: + raise click.ClickException(str(error)) from None + + if __name__ == "__main__": cli() diff --git a/in2lambda/source/__init__.py b/in2lambda/source/__init__.py new file mode 100644 index 0000000..33aef4f --- /dev/null +++ b/in2lambda/source/__init__.py @@ -0,0 +1,431 @@ +"""Freezes a source document, so that its text can be quoted by line range. + +Anything that writes questions from a document - the in2lambda agent, say - needs to +take the wording out of the source rather than retype it, and a line range is only an +address if the text it points into cannot move underneath it. So the document is frozen +once: converted to markdown, hashed, and written down beside a ``draft.json`` listing +every top-level block with the lines it spans. + +Everything here needs pandoc, and the parsing needs panflute, which only the ``convert`` +extra installs; :func:`add` says so rather than failing on the import. +""" + +import hashlib +import importlib.util +import json +import re +import shutil +import subprocess +from dataclasses import dataclass +from pathlib import Path +from typing import Any + +DRAFT = "draft.json" +"""What a frozen source is written to, beside the source itself.""" + +_FIELDS = ("source", "hash", "blocks") +"""What a draft has in it, and so what one has to have for anything here to read it.""" + +_MARKDOWN = "commonmark_x" +"""The dialect the frozen markdown is written in, and read back as. + +Writer and reader have to agree: pandoc's ``markdown`` writer emits fenced divs and +bracketed spans that a commonmark reader would take as ordinary text. ``commonmark_x`` +also covers the ``$...$`` maths and the ``{width=...}`` attributes a converted document +carries. +""" + +_POSITION = re.compile(r"(?:[^@;]*@)?(\d+):\d+-(\d+):(\d+)") +"""One ``line:column-line:column`` of a ``data-pos``, which may name a file and repeat.""" + + +class SourceError(RuntimeError): + """Freezing or printing a source could not be done, for a reason worth printing. + + The command line turns any of these into a message and a non-zero exit, so + anything a reader could do something about - a draft from somewhere else, a file + that has moved - is raised as one of these rather than left as whatever the + standard library raised on the way past. + """ + + +class ConversionToolsMissing(SourceError): + """Document conversion was asked for without pandoc or panflute installed.""" + + +class DraftExists(SourceError): + """A draft is already there and was not made from this version of the source.""" + + +class DraftMissing(SourceError): + """There is no draft to show in the directory asked about.""" + + +class DraftUnreadable(SourceError): + """There is a file where the draft goes, but it is not a draft.""" + + +class SourceUnreadable(SourceError): + """The markdown to read has moved, or is not text.""" + + +def _require_conversion_tools() -> None: + missing = [] + if shutil.which("pandoc") is None: + missing.append("pandoc (see https://pandoc.org/installing.html)") + if importlib.util.find_spec("panflute") is None: + missing.append("panflute (pip install 'in2lambda[convert]')") + if missing: + raise ConversionToolsMissing( + f"Converting documents needs {' and '.join(missing)}." + ) + + +def file_type(file: str) -> str: + """Determines which pandoc file format to use for a given file. + + See https://github.com/jgm/pandoc/blob/bad922a69236e22b20d51c4ec0b90c5a6c038433/src/Text/Pandoc/Format.hs#L171 + (or any newer commit) for pandoc's supported file extensions. + + Args: + file: A file path with the file extension included. + + Returns: + An option in `pandoc --list-input-formats` that matches the given file type + + Examples: + >>> from in2lambda.source import file_type + >>> file_type("example.tex") + 'latex' + >>> file_type("/some/random/path/demo.md") + 'markdown' + >>> file_type("no_extension") + Traceback (most recent call last): + RuntimeError: Unsupported file extension: .no_extension + >>> file_type("demo.unknown_extension") + Traceback (most recent call last): + RuntimeError: Unsupported file extension: .unknown_extension + """ + match (extension := file.split(".")[-1].lower()): + case "tex" | "latex" | "ltx": + return "latex" + case ( + "md" + | "rmd" + | "markdown" + | "mdown" + | "mdwn" + | "mkd" + | "mkdn" + | "text" + | "txt" + ): + return "markdown" + case "docx": + return "docx" # Pandoc doesn't seem to support .doc, and panflute doesn't like .docx. + raise RuntimeError(f"Unsupported file extension: .{extension}") + + +def _pandoc(file: str, to: str) -> bytes: + """The given file, as pandoc writes it in the `to` format. + + Undecoded, because what is written to disk and what is hashed have to be the same + bytes; whoever wants the text of it decodes it themselves. + """ + return subprocess.check_output(["pandoc", file, "-f", file_type(file), "-t", to]) + + +def _digest(data: bytes) -> str: + """How a frozen markdown is named in its draft, so that a change to it shows up. + + The bytes of the file, not the text they decode to: the draft is checked by whoever + is quoting the markdown, who has nothing but the file, and `sha256sum` on it has to + give the same answer whatever the line endings in it are. + """ + return f"sha256:{hashlib.sha256(data).hexdigest()}" + + +def _source(path: Path) -> tuple[bytes, str]: + """A markdown file as bytes and as text, given it is still there and still text. + + Both freezing and showing read one, and someone who has moved the file or saved it + in some other encoding wants telling which it was, not a traceback. The bytes are + what gets hashed, and `bytes.decode` rewrites no line endings, so the text still + has whatever the file has. + """ + try: + raw = path.read_bytes() + except FileNotFoundError: + raise SourceUnreadable( + f"There is no {path}. Put it back, or freeze the document it came from " + "again with in2lambda source add --start-over." + ) from None + try: + return raw, raw.decode("utf-8") + except UnicodeDecodeError: + raise SourceUnreadable( + f"{path} is not UTF-8 text, so it cannot be read as markdown. Save it as " + "UTF-8 and try again." + ) from None + + +def _draft(path: Path) -> dict[str, Any]: + """The draft at the given path, given that something here wrote it. + + Raises: + DraftMissing: nothing is there at all. + DraftUnreadable: something is, but it is not JSON or it is not a draft. Either + way it is not this package's to read from or write over. + """ + if not path.is_file(): + raise DraftMissing( + f"There is no {DRAFT} in {path.parent.resolve()}. " + "Run in2lambda source add FILE first." + ) + advice = ( + "Move it aside and run in2lambda source add FILE, or pass --start-over to " + "write over it." + ) + try: + draft = json.loads(path.read_text(encoding="utf-8")) + fields = draft.keys() + except (ValueError, AttributeError) as error: + # AttributeError: valid JSON, but a list or a number rather than an object. + raise DraftUnreadable( + f"{path} cannot be read as a draft: {error}. {advice}" + ) from None + if missing := [field for field in _FIELDS if field not in fields]: + raise DraftUnreadable( + f"{path} is not a draft anything here wrote: it has no " + f"{' or '.join(missing)} in it. {advice}" + ) + return draft + + +@dataclass +class Block: + """One top-level block of a frozen source, and the lines it spans. + + Lines are 1-based and inclusive, so `start` and `end` are the numbers + :func:`show` prints beside that block's first and last line. + """ + + id: str + type: str + start: int + end: int + + def to_dict(self) -> dict[str, str | int]: + """The block as it is written into ``draft.json``.""" + return {"id": self.id, "type": self.type, "start": self.start, "end": self.end} + + +def blocks(markdown: str) -> list[Block]: + r"""Every top-level block of some markdown, in the order it is written. + + Args: + markdown: A document in the dialect :func:`add` freezes to. + + Returns: + One :class:`Block` per block, numbered ``b1`` onwards. The blocks do not + overlap and every line of the document falls in at most one: a block that is + none of the types the agent quotes is still listed, as ``other``, rather than + leaving its lines unaddressable. + + Examples: + >>> from in2lambda.source import blocks + >>> blocks("# Title\n\nSome words.\n") + [Block(id='b1', type='heading', start=1, end=1), Block(id='b2', type='paragraph', start=3, end=3)] + """ + import panflute as pf + + document = pf.convert_text( + markdown, input_format=f"{_MARKDOWN}+sourcepos", standalone=True + ) + found = [span for element in document.content for span in _spans(element, pf)] + # Where no blank line separates one block from the next - a list straight after a + # paragraph, a definition list - pandoc reports the first as running on into the + # second's first line, so no block is allowed to reach where the next one starts, + # nor past the end of the document. + limits = [start - 1 for _, start, _ in found[1:]] + [len(markdown.splitlines())] + return [ + Block(f"b{number}", kind, start, min(end, limit)) + for number, ((kind, start, end), limit) in enumerate(zip(found, limits), 1) + ] + + +def _spans(element, pf): # type: ignore[no-untyped-def] + """The ``(type, start, end)`` triples one top-level element accounts for. + + A list is several: the ticket asks for a list item, not a list, and an item spans + everything nested under it. + """ + inner = _unwrapped(element, pf) + if isinstance(inner, (pf.BulletList, pf.OrderedList)): + # An item with nothing in it - a lone bullet, which a .docx often has - holds + # no element to take a position from, so there is no range to give it and it + # is left out rather than guessed at. + return [ + ("list item", _range(item.content[0])[0], _range(item.content[-1])[1]) + for item in inner.content + if len(item.content) + ] + return [(_kind(inner, pf), *_range(element))] + + +def _unwrapped(element, pf): # type: ignore[no-untyped-def] + """What an element is, past the Div that `sourcepos` wraps it in. + + Only elements that take attributes of their own (a heading, a table) carry + ``data-pos`` directly; pandoc wraps the rest in a Div to hang it on. + """ + if isinstance(element, pf.Div) and element.attributes.get("wrapper"): + return element.content[0] + return element + + +def _kind(inner, pf) -> str: # type: ignore[no-untyped-def] + """Which of the ticket's block types an unwrapped element is.""" + if isinstance(inner, pf.Header): + return "heading" + if isinstance(inner, (pf.Para, pf.Plain)): + # A paragraph holding nothing but one image, or one $$...$$, is that thing. + contents = [ + item + for element in inner.content + for item in (element.content if isinstance(element, pf.Span) else [element]) + if not isinstance(item, (pf.Space, pf.SoftBreak)) + ] + if len(contents) == 1: + if isinstance(contents[0], pf.Math) and contents[0].format == "DisplayMath": + return "display maths" + if isinstance(contents[0], pf.Image): + return "image" + return "paragraph" + return "other" + + +def _range(element) -> tuple[int, int]: # type: ignore[no-untyped-def] + """The first and last line an element covers, from its ``data-pos``. + + An element may carry more than one position, in which case they are parts of it and + the whole of it is wanted. An end at column 1 means the block stopped before that + line, which is how pandoc reports every block that ends in a newline. + """ + positions = _POSITION.findall(element.attributes["data-pos"]) + return ( + min(int(start) for start, _, _ in positions), + max( + int(end) - 1 if column == "1" else int(end) for _, end, column in positions + ), + ) + + +def add(file: str, start_over: bool = False) -> Path: + """Freezes a document and writes the draft of it beside the file. + + A .docx or .tex file is converted to markdown next to it; a markdown file is taken + as it is and nothing is copied. Either way the markdown is hashed and its blocks + written to ``draft.json``, so that whatever quotes the source by line range can tell + that the lines it was given still say what they said. + + Args: + file: The document to freeze, as .docx, .tex or markdown. + start_over: Freeze the file again, discarding whatever is already there. + + Returns: + The path of the ``draft.json`` that was written. + + Raises: + ConversionToolsMissing: pandoc or panflute is not installed. + SourceUnreadable: the file is markdown, but not UTF-8 text. + DraftUnreadable: there is a draft.json beside the file that nothing here wrote, + so it is not ours to read a hash out of or to write over. + DraftExists: the source has changed since it was frozen, or the markdown would + overwrite a file that no draft claims. Neither happens with `start_over`. + """ + _require_conversion_tools() + source = Path(file) + if file_type(file) == "markdown": + raw, markdown = _source(source) + frozen = source + else: + raw = _pandoc(file, _MARKDOWN) + markdown = raw.decode("utf-8") + frozen = source.with_suffix(".md") + draft = source.parent / DRAFT + digest = _digest(raw) + + if not start_over: + if draft.is_file(): + if _draft(draft)["hash"] != digest: + raise DraftExists( + f"{source.name} has changed since {DRAFT} was written from it. " + "Run in2lambda source add --start-over to freeze it again, which " + "invalidates every line range taken from the old draft." + ) + elif frozen != source and frozen.exists(): + raise DraftExists( + f"{frozen.name} is already there and no {DRAFT} claims it, so it is " + "not ours to overwrite. Move it aside, or run in2lambda source add " + "--start-over." + ) + + # Before either file is written: a parse that fails half way through would + # otherwise leave the markdown there with no draft claiming it, and the next run + # would refuse to touch a file this one wrote. + found = [block.to_dict() for block in blocks(markdown)] + + if frozen != source: + # The bytes pandoc wrote, so that the file on disk is what `digest` is of; + # writing text would rewrite the line endings on Windows and it would not be. + frozen.write_bytes(raw) + draft.write_text( + json.dumps( + {"source": frozen.name, "hash": digest, "blocks": found}, + indent=2, + ) + + "\n", + encoding="utf-8", + ) + return draft + + +def show(directory: str = ".") -> str: + """The frozen markdown of a draft, numbered, with block ids in the margin. + + Args: + directory: Where the ``draft.json`` to print is. + + Returns: + One line per line of the frozen markdown: the id of the block starting there, + where one does, then the line number and the line itself. + + Raises: + DraftMissing: there is no draft in that directory. + DraftUnreadable: what is there is not a draft anything here wrote. + SourceUnreadable: the markdown the draft names has moved, or is not text. + DraftExists: the markdown has changed since the draft was written from it, so + the ids would be printed against lines they are not the ids of. + """ + draft_path = Path(directory) / DRAFT + draft = _draft(draft_path) + raw, markdown = _source(draft_path.parent / draft["source"]) + # A line range is only an address while the lines have not moved: printing ids + # against markdown the draft was not written from would be worse than printing + # nothing, because it would look right. + if _digest(raw) != draft["hash"]: + raise DraftExists( + f"{draft['source']} has changed since {DRAFT} was written from it, so its " + "block ids no longer name the lines they were written against. Run " + "in2lambda source add --start-over to freeze the file as it now is." + ) + + ids = {block["start"]: block["id"] for block in draft["blocks"]} + lines = markdown.splitlines() + margin = max((len(block_id) for block_id in ids.values()), default=0) + numbers = len(str(len(lines))) + return "\n".join( + f"{ids.get(number, ''):>{margin}} {number:>{numbers}} {line}".rstrip() + for number, line in enumerate(lines, start=1) + ) diff --git a/tests/conftest.py b/tests/conftest.py index 7e50eba..189112d 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -19,6 +19,12 @@ PROBLEM_SETS = sorted(path for path in PROBLEMS_DIR.iterdir() if path.is_dir()) """Every folder of the above, found the same way: covering a check means adding one.""" +SOURCES_DIR = Path(__file__).parent / "fixtures" / "sources" +"""One document per folder for `in2lambda source add`, beside the blocks it should find.""" + +SOURCES = sorted(path for path in SOURCES_DIR.iterdir() if path.is_dir()) +"""Every source folder, so that covering another construct is a folder and no code.""" + @pytest.fixture(scope="session") def filters_dir() -> str: diff --git a/tests/fixtures/sources/README.md b/tests/fixtures/sources/README.md new file mode 100644 index 0000000..0d17e9a --- /dev/null +++ b/tests/fixtures/sources/README.md @@ -0,0 +1,28 @@ +# Sources to freeze + +Each folder here is one document for `in2lambda source add` to freeze, beside the block list +it should write into `draft.json`. `markdown`, `tex` and `docx` say the same thing in the three +formats the command takes, so that what a heading or a list item comes out as does not depend on +which format an author brought it in; `empty_list_item` is a bullet with nothing in it, which has +no position of its own and so no block; `unseparated_list` is a list with no blank line before it, +which pandoc reports as part of the paragraph above, so that paragraph's block has to stop where +the list starts rather than where pandoc says it ends; `crlf` is the `markdown` case saved with +Windows line endings, which is what a document off a teacher's machine usually has, and it has to +freeze to the same blocks and to a hash that `sha256sum source.md` reproduces. The `.gitattributes` +at the top of the repository is what stops a checkout rewriting those endings away. + +To cover another construct, add a folder: one `source.md`, `source.tex` or `source.docx`, and +the `expected.json` the test compares the draft's `blocks` against. + +The line ranges of the `.tex` and `.docx` cases are ranges in the markdown pandoc writes, not +in the document itself, so they move if pandoc's `commonmark_x` writer changes. They were +produced with **pandoc 3.9.0.2**. + +`docx/source.docx` was made from `markdown/source.md` with `pandoc source.md -o source.docx`, +run beside a `figure.png` so that the image is embedded rather than dropped, and with the +image's alt text removed: pandoc turns a captioned image into a figure, which `commonmark_x` +can only write as raw HTML, and the block would then be `other` rather than `image`. A Word +image usually has no alt text, so this is also the ordinary case. + +The image the docx embeds is referenced as `media/rId9.png`, which is not extracted - nothing +here reads the image, only the lines around it. diff --git a/tests/fixtures/sources/crlf/expected.json b/tests/fixtures/sources/crlf/expected.json new file mode 100644 index 0000000..f190bf3 --- /dev/null +++ b/tests/fixtures/sources/crlf/expected.json @@ -0,0 +1,44 @@ +[ + { + "end": 1, + "id": "b1", + "start": 1, + "type": "heading" + }, + { + "end": 3, + "id": "b2", + "start": 3, + "type": "paragraph" + }, + { + "end": 5, + "id": "b3", + "start": 5, + "type": "list item" + }, + { + "end": 6, + "id": "b4", + "start": 6, + "type": "list item" + }, + { + "end": 10, + "id": "b5", + "start": 8, + "type": "display maths" + }, + { + "end": 12, + "id": "b6", + "start": 12, + "type": "image" + }, + { + "end": 16, + "id": "b7", + "start": 14, + "type": "other" + } +] diff --git a/tests/fixtures/sources/crlf/source.md b/tests/fixtures/sources/crlf/source.md new file mode 100644 index 0000000..a5c40ba --- /dev/null +++ b/tests/fixtures/sources/crlf/source.md @@ -0,0 +1,16 @@ +# Hydraulic scale + +A hydraulic scale has two pistons joined by oil. Find the load the large one carries. + +- The small piston has area $a$. +- The large piston has area $A$. + +$$ +F = p A +$$ + +![The two pistons](figure.png) + +| Quantity | Symbol | +|----------|--------| +| Area | $A$ | diff --git a/tests/fixtures/sources/docx/expected.json b/tests/fixtures/sources/docx/expected.json new file mode 100644 index 0000000..3b34c33 --- /dev/null +++ b/tests/fixtures/sources/docx/expected.json @@ -0,0 +1,44 @@ +[ + { + "end": 1, + "id": "b1", + "start": 1, + "type": "heading" + }, + { + "end": 4, + "id": "b2", + "start": 3, + "type": "paragraph" + }, + { + "end": 6, + "id": "b3", + "start": 6, + "type": "list item" + }, + { + "end": 7, + "id": "b4", + "start": 7, + "type": "list item" + }, + { + "end": 9, + "id": "b5", + "start": 9, + "type": "display maths" + }, + { + "end": 12, + "id": "b6", + "start": 11, + "type": "image" + }, + { + "end": 16, + "id": "b7", + "start": 14, + "type": "other" + } +] diff --git a/tests/fixtures/sources/docx/source.docx b/tests/fixtures/sources/docx/source.docx new file mode 100644 index 0000000..b03ab27 Binary files /dev/null and b/tests/fixtures/sources/docx/source.docx differ diff --git a/tests/fixtures/sources/empty_list_item/expected.json b/tests/fixtures/sources/empty_list_item/expected.json new file mode 100644 index 0000000..4b98e69 --- /dev/null +++ b/tests/fixtures/sources/empty_list_item/expected.json @@ -0,0 +1,8 @@ +[ + { + "end": 2, + "id": "b1", + "start": 2, + "type": "list item" + } +] diff --git a/tests/fixtures/sources/empty_list_item/source.md b/tests/fixtures/sources/empty_list_item/source.md new file mode 100644 index 0000000..33c7283 --- /dev/null +++ b/tests/fixtures/sources/empty_list_item/source.md @@ -0,0 +1,2 @@ +- +- The large piston has area $A$. diff --git a/tests/fixtures/sources/markdown/expected.json b/tests/fixtures/sources/markdown/expected.json new file mode 100644 index 0000000..f190bf3 --- /dev/null +++ b/tests/fixtures/sources/markdown/expected.json @@ -0,0 +1,44 @@ +[ + { + "end": 1, + "id": "b1", + "start": 1, + "type": "heading" + }, + { + "end": 3, + "id": "b2", + "start": 3, + "type": "paragraph" + }, + { + "end": 5, + "id": "b3", + "start": 5, + "type": "list item" + }, + { + "end": 6, + "id": "b4", + "start": 6, + "type": "list item" + }, + { + "end": 10, + "id": "b5", + "start": 8, + "type": "display maths" + }, + { + "end": 12, + "id": "b6", + "start": 12, + "type": "image" + }, + { + "end": 16, + "id": "b7", + "start": 14, + "type": "other" + } +] diff --git a/tests/fixtures/sources/markdown/source.md b/tests/fixtures/sources/markdown/source.md new file mode 100644 index 0000000..c5c9769 --- /dev/null +++ b/tests/fixtures/sources/markdown/source.md @@ -0,0 +1,16 @@ +# Hydraulic scale + +A hydraulic scale has two pistons joined by oil. Find the load the large one carries. + +- The small piston has area $a$. +- The large piston has area $A$. + +$$ +F = p A +$$ + +![The two pistons](figure.png) + +| Quantity | Symbol | +|----------|--------| +| Area | $A$ | diff --git a/tests/fixtures/sources/tex/expected.json b/tests/fixtures/sources/tex/expected.json new file mode 100644 index 0000000..f4a5339 --- /dev/null +++ b/tests/fixtures/sources/tex/expected.json @@ -0,0 +1,38 @@ +[ + { + "end": 1, + "id": "b1", + "start": 1, + "type": "heading" + }, + { + "end": 4, + "id": "b2", + "start": 3, + "type": "paragraph" + }, + { + "end": 6, + "id": "b3", + "start": 6, + "type": "list item" + }, + { + "end": 8, + "id": "b4", + "start": 8, + "type": "list item" + }, + { + "end": 10, + "id": "b5", + "start": 10, + "type": "display maths" + }, + { + "end": 12, + "id": "b6", + "start": 12, + "type": "image" + } +] diff --git a/tests/fixtures/sources/tex/source.tex b/tests/fixtures/sources/tex/source.tex new file mode 100644 index 0000000..9d5b56f --- /dev/null +++ b/tests/fixtures/sources/tex/source.tex @@ -0,0 +1,14 @@ +\section{Hydraulic scale} + +A hydraulic scale has two pistons joined by oil. Find the load the large one carries. + +\begin{itemize} + \item The small piston has area $a$. + \item The large piston has area $A$. +\end{itemize} + +\[ +F = p A +\] + +\includegraphics{figure.png} diff --git a/tests/fixtures/sources/unseparated_list/expected.json b/tests/fixtures/sources/unseparated_list/expected.json new file mode 100644 index 0000000..3f96095 --- /dev/null +++ b/tests/fixtures/sources/unseparated_list/expected.json @@ -0,0 +1,26 @@ +[ + { + "end": 1, + "id": "b1", + "start": 1, + "type": "heading" + }, + { + "end": 3, + "id": "b2", + "start": 3, + "type": "paragraph" + }, + { + "end": 4, + "id": "b3", + "start": 4, + "type": "list item" + }, + { + "end": 5, + "id": "b4", + "start": 5, + "type": "list item" + } +] diff --git a/tests/fixtures/sources/unseparated_list/source.md b/tests/fixtures/sources/unseparated_list/source.md new file mode 100644 index 0000000..4e26a55 --- /dev/null +++ b/tests/fixtures/sources/unseparated_list/source.md @@ -0,0 +1,5 @@ +# Hydraulic scale + +A hydraulic scale has two pistons joined by oil. +- The small piston has area $a$. +- The large piston has area $A$. diff --git a/tests/test_cli.py b/tests/test_cli.py index a8b96ea..047f2a1 100644 --- a/tests/test_cli.py +++ b/tests/test_cli.py @@ -85,4 +85,4 @@ def test_completing_the_old_form_offers_the_subcommand() -> None: cli, {}, "in2lambda", "_IN2LAMBDA_COMPLETE" ).get_completions(["./questions.tex"], "") - assert [candidate.value for candidate in completions] == ["convert"] + assert [candidate.value for candidate in completions] == ["convert", "source"] diff --git a/tests/test_conversion_tools.py b/tests/test_conversion_tools.py index 26c1bc4..2722f9c 100644 --- a/tests/test_conversion_tools.py +++ b/tests/test_conversion_tools.py @@ -46,3 +46,15 @@ def test_cli_exits_with_message(filters_dir: str, monkeypatch, tmp_path) -> None assert result.exit_code != 0 assert PANFLUTE_HINT in result.output assert isinstance(result.exception, SystemExit) + + +def test_source_add_exits_with_message(filters_dir: str, monkeypatch) -> None: + """Freezing a document needs the same tools, and says so the same way.""" + monkeypatch.setitem(sys.modules, "panflute", None) + example = os.path.join(filters_dir, "PartsSepSol", "example.tex") + + result = CliRunner().invoke(cli, ["source", "add", example]) + + assert result.exit_code != 0 + assert PANFLUTE_HINT in result.output + assert isinstance(result.exception, SystemExit) diff --git a/tests/test_source.py b/tests/test_source.py new file mode 100644 index 0000000..14fe82a --- /dev/null +++ b/tests/test_source.py @@ -0,0 +1,208 @@ +"""Freezing a source document, and what happens when it changes afterwards. + +Each folder in ``fixtures/sources`` is one document beside the block list freezing it +should produce, so covering another construct means adding a folder rather than a test. +The rest is what the command line does - printing a draft, and refusing one whose source +has moved on or which nothing here wrote - which is not something a fixture can say. +""" + +import hashlib +import json +import re +import shutil +from pathlib import Path + +import pytest +from click.testing import CliRunner +from conftest import SOURCES, SOURCES_DIR + +from in2lambda.main import cli + +MARKDOWN = SOURCES_DIR / "markdown" +"""The case the tests below happen to use; what they check holds for any of them.""" + + +def _frozen(directory: Path) -> Path: + """The document a folder holds, whatever format it is in.""" + (source,) = (path for path in directory.iterdir() if path.stem == "source") + return source + + +@pytest.mark.parametrize("folder", SOURCES, ids=lambda path: path.name) +def test_source_add_finds_the_expected_blocks(folder: Path, tmp_path: Path) -> None: + """Each document freezes to the block list written beside it, hash and all.""" + shutil.copytree(folder, tmp_path, dirs_exist_ok=True) + + result = CliRunner().invoke(cli, ["source", "add", str(_frozen(tmp_path))]) + + assert result.exit_code == 0, result.output + draft = json.loads((tmp_path / "draft.json").read_text()) + assert draft["blocks"] == json.loads((folder / "expected.json").read_text()) + markdown = (tmp_path / draft["source"]).read_bytes() + assert draft["hash"] == f"sha256:{hashlib.sha256(markdown).hexdigest()}" + + +def test_freezing_again_is_refused_once_the_source_has_changed( + tmp_path: Path, monkeypatch +) -> None: + """A draft is only an address for line ranges while the lines have not moved.""" + monkeypatch.setenv("COLUMNS", "200") # So the message is not wrapped mid-sentence. + source = tmp_path / "source.md" + shutil.copy(_frozen(MARKDOWN), source) + runner = CliRunner() + assert runner.invoke(cli, ["source", "add", str(source)]).exit_code == 0 + frozen = (tmp_path / "draft.json").read_text() + + # Freezing the same file again changes nothing, so it is allowed. + assert runner.invoke(cli, ["source", "add", str(source)]).exit_code == 0 + assert (tmp_path / "draft.json").read_text() == frozen + + source.write_text(f"{source.read_text()}\nAn afterthought.\n") + result = runner.invoke(cli, ["source", "add", str(source)]) + + assert result.exit_code != 0 + assert "--start-over" in result.output + assert (tmp_path / "draft.json").read_text() == frozen + + assert ( + runner.invoke(cli, ["source", "add", str(source), "--start-over"]).exit_code + == 0 + ) + assert (tmp_path / "draft.json").read_text() != frozen + + +def test_a_markdown_file_no_draft_claims_is_not_overwritten( + tmp_path: Path, monkeypatch +) -> None: + """Freezing a .tex writes a .md beside it, which may be someone else's work.""" + monkeypatch.setenv("COLUMNS", "200") + shutil.copy(_frozen(SOURCES_DIR / "tex"), tmp_path / "source.tex") + theirs = "# Notes I wrote by hand\n" + (tmp_path / "source.md").write_text(theirs) + runner = CliRunner() + + result = runner.invoke(cli, ["source", "add", str(tmp_path / "source.tex")]) + + assert result.exit_code != 0 + assert "--start-over" in result.output + assert (tmp_path / "source.md").read_text() == theirs + assert not (tmp_path / "draft.json").exists() + + # Saying to start over is saying to overwrite it. + result = runner.invoke( + cli, ["source", "add", str(tmp_path / "source.tex"), "--start-over"] + ) + + assert result.exit_code == 0, result.output + assert (tmp_path / "source.md").read_text() != theirs + assert json.loads((tmp_path / "draft.json").read_text())["source"] == "source.md" + + +def test_source_show_numbers_the_lines_and_names_the_blocks( + tmp_path: Path, monkeypatch +) -> None: + """`source show` is how someone checks the ids the agent will be quoting.""" + shutil.copytree(MARKDOWN, tmp_path, dirs_exist_ok=True) + monkeypatch.chdir(tmp_path) + runner = CliRunner() + runner.invoke(cli, ["source", "add", str(_frozen(tmp_path))]) + + result = runner.invoke(cli, ["source", "show"]) + + assert result.exit_code == 0, result.output + lines = result.output.splitlines() + markdown = (tmp_path / "source.md").read_text().splitlines() + assert lines[0] == f"b1 1 {markdown[0]}" + assert lines[1] == " 2" # A blank line still gets its number, with no id. + assert len(lines) == len(markdown) + + # Every block's id sits on the line it starts at, and nothing else carries one. + blocks = json.loads((tmp_path / "draft.json").read_text())["blocks"] + for block in blocks: + assert lines[block["start"] - 1].split()[0] == block["id"] + assert sum(bool(re.match(r" *b\d+ ", line)) for line in lines) == len(blocks) + + +def test_source_show_without_a_draft_says_so(tmp_path: Path, monkeypatch) -> None: + """Running it in the wrong directory is a message, not a traceback.""" + monkeypatch.chdir(tmp_path) + + result = CliRunner().invoke(cli, ["source", "show"]) + + assert result.exit_code != 0 + assert "in2lambda source add" in result.output + assert isinstance(result.exception, SystemExit) + + +def test_source_show_refuses_once_the_source_has_changed( + tmp_path: Path, monkeypatch +) -> None: + """Ids printed against lines they are not the ids of would look right and be wrong.""" + monkeypatch.setenv("COLUMNS", "200") + shutil.copytree(MARKDOWN, tmp_path, dirs_exist_ok=True) + monkeypatch.chdir(tmp_path) + runner = CliRunner() + assert runner.invoke(cli, ["source", "add", "source.md"]).exit_code == 0 + + source = tmp_path / "source.md" + source.write_text(f"An afterthought.\n\n{source.read_text()}") + result = runner.invoke(cli, ["source", "show"]) + + assert result.exit_code != 0 + assert "--start-over" in result.output + assert isinstance(result.exception, SystemExit) + + +@pytest.mark.parametrize( + "content", ["{ not json at all", '{"blocks": []}'], ids=["not-json", "foreign"] +) +@pytest.mark.parametrize( + "arguments", + [["source", "add", "source.md"], ["source", "show"]], + ids=["add", "show"], +) +def test_a_draft_from_somewhere_else_is_refused( + content: str, arguments: list[str], tmp_path: Path, monkeypatch +) -> None: + """A draft.json nothing here wrote is neither read from nor written over.""" + monkeypatch.setenv("COLUMNS", "200") + shutil.copy(_frozen(MARKDOWN), tmp_path / "source.md") + (tmp_path / "draft.json").write_text(content) + monkeypatch.chdir(tmp_path) + + result = CliRunner().invoke(cli, arguments) + + assert result.exit_code != 0 + assert "--start-over" in result.output + assert isinstance(result.exception, SystemExit) + assert (tmp_path / "draft.json").read_text() == content + + +def test_a_frozen_file_that_has_gone_is_a_message(tmp_path: Path, monkeypatch) -> None: + """The draft names the markdown, and someone may well have moved it since.""" + monkeypatch.setenv("COLUMNS", "200") + shutil.copytree(MARKDOWN, tmp_path, dirs_exist_ok=True) + monkeypatch.chdir(tmp_path) + runner = CliRunner() + assert runner.invoke(cli, ["source", "add", "source.md"]).exit_code == 0 + (tmp_path / "source.md").unlink() + + result = runner.invoke(cli, ["source", "show"]) + + assert result.exit_code != 0 + assert "source.md" in result.output + assert isinstance(result.exception, SystemExit) + + +def test_a_source_that_is_not_text_is_a_message(tmp_path: Path, monkeypatch) -> None: + """A .md saved in some other encoding cannot be read as markdown, and says so.""" + monkeypatch.setenv("COLUMNS", "200") + source = tmp_path / "source.md" + source.write_bytes(b"\xff\xfe# Hydraulic scale\n") + + result = CliRunner().invoke(cli, ["source", "add", str(source)]) + + assert result.exit_code != 0 + assert "UTF-8" in result.output + assert isinstance(result.exception, SystemExit) + assert not (tmp_path / "draft.json").exists()