diff --git a/README.md b/README.md index d979b9e..b02084f 100644 --- a/README.md +++ b/README.md @@ -197,8 +197,8 @@ pdf` runs the PDFs too, which needs Mathpix credentials and one call per PDF. A set is a folder holding at least one questions document. Each set converts as a folder run of `convert` does: one model call writes the set's filter from the first -sheet of it pandoc can read, and each sheet of the set then runs through route A and -that filter. A sheet and the solutions file beside it are one conversion and one row, +sheet of it pandoc reads itself, or from the first PDF where the set has no other, and +each sheet of the set then runs through route A and that filter. A sheet and the solutions file beside it are one conversion and one row, named after the questions file. A folder of figures, a tex drawing with no `\begin{document}` among them, and a folder holding a solutions file alone, are not sets and have no row. @@ -217,11 +217,12 @@ tokens, seconds, reason `reason` is empty where the sheet ran through both routes and built its set. One sheet whose conversion raises is one row, with `no set:` and the -error as its reason, and the sheets after it still run. A set with no filter — its -filter call did not finish, or every sheet of it is a PDF, which pandoc cannot read — -converts every sheet through route A alone, and each of those rows reads `no filter:` -and why. A PDF among sheets pandoc reads leaves the rest of the set on both routes: the -filter is written from one of those sheets, and the PDF alone fails route B. +error as its reason, and the sheets after it still run. A set whose filter call did not +finish converts every sheet through route A alone, and each of those rows reads `no +filter:` and why. A PDF runs route B over the markdown its OCR made, which is the +markdown route A reads, so a set of PDFs has a filter like any other; where the filter +was written from a tex or docx sheet beside it, that filter may still fail on the PDF, +and that row alone reads `route B failed:`. `corpus` exits 0 where it ran at least one sheet and every sheet built a set. It exits 1 where it found no sheet, and where any sheet built no set — a row whose reason begins @@ -311,9 +312,8 @@ Those are the only calls a saved target spares. A target with a filter runs rout on every run, and a model adjudicates every field the two routes word differently. A verdict can go the other way on a later run, so the wording of a difference and the `flagged` count move from run to run while the fields `differs.txt` accepts stay -accepted. A target whose questions document is a PDF has no filter: -pandoc cannot read a PDF, so route B does not run and route A converts the pages' -markdown alone. `--out` (default `./out`) is where each target's set is written, under +accepted. A target whose questions document is a PDF has no filter: route B does not +run, and route A converts the pages' markdown alone. `--out` (default `./out`) is where each target's set is written, under the target's own name, and `--cache` (default `./.in2lambda-agent`) is where the OCR of each PDF is kept. diff --git a/in2lambda_agent/cli.py b/in2lambda_agent/cli.py index f98f9cc..f33645a 100644 --- a/in2lambda_agent/cli.py +++ b/in2lambda_agent/cli.py @@ -277,7 +277,10 @@ def convert_command(args: argparse.Namespace) -> int: out_dir.mkdir(parents=True, exist_ok=True) lua = out_dir / "filter.lua" lua.write_text( - routes.write_filter(document, solutions, backend)[0], encoding="utf-8" + routes.write_filter( + document, solutions, backend, cache_dir=args.cache, settings=settings + )[0], + encoding="utf-8", ) result = routes.convert( document, diff --git a/in2lambda_agent/routes.py b/in2lambda_agent/routes.py index 74aa9b6..1992dc6 100644 --- a/in2lambda_agent/routes.py +++ b/in2lambda_agent/routes.py @@ -489,14 +489,33 @@ def _areas_counted(areas: dict[str, list[dict[str, Any]]]) -> str: _UNDERLINE = Path(__file__).parent / "underline.lua" +def _ocr(document: Path, cache_dir: Path, settings: Settings): + """The OCR of a PDF, from the cache where it holds one.""" + from in2lambda_agent.mathpix import MathpixClient + from in2lambda_agent.ocr import ocr_pdf + + return ocr_pdf(document, cache_dir=cache_dir, client=MathpixClient.from_settings(settings)) + + +def pandoc_reads(document: Path, cache_dir: Path, settings: Settings) -> Path: + """The file pandoc is given for a document: for a PDF, its OCR markdown. + + Pandoc reads no PDF, so both halves of route B - the tree the filter is + written from, and the run of that filter - read the markdown Mathpix made of + the pages, which is what route A reads too. Every other document pandoc + reads itself, and no credential is asked for. + """ + document = Path(document) + if document.suffix.lower() == ".pdf": + return _ocr(document, cache_dir, settings).markdown + return document + + def markdown_of(document: Path, cache_dir: Path, settings: Settings) -> tuple[str, Path]: """The document as markdown, and the folder its images are in.""" document = Path(document) if document.suffix.lower() == ".pdf": - from in2lambda_agent.mathpix import MathpixClient - from in2lambda_agent.ocr import ocr_pdf - - ocr = ocr_pdf(document, cache_dir=cache_dir, client=MathpixClient.from_settings(settings)) + ocr = _ocr(document, cache_dir, settings) return ocr.markdown.read_text(encoding="utf-8"), ocr.markdown.parent if document.suffix.lower() in (".md", ".markdown"): return document.read_text(encoding="utf-8"), document.parent @@ -545,7 +564,8 @@ def convert( """Route A, route B where a filter is given, reconcile, verify, write. Route B reads the solutions document too, under its own role, and the two runs are - merged before the comparison. Where a filter run fails, the route A reply is the + merged before the comparison. Of a PDF it reads the markdown the OCR made, which is + what route A reads. Where a filter run fails, the route A reply is the result and `route_b_error` holds pandoc's message, so that one sheet of a folder does not stop the other eight. @@ -601,9 +621,11 @@ def said(stage: str, message: str) -> None: said("route B", "did not run: no filter") else: try: - other = run_filter(lua, document) + # A PDF's filter runs over the markdown its OCR made, which + # `markdown_of` has by now put in the cache. + other = run_filter(lua, pandoc_reads(document, cache_dir, settings)) if solutions is not None: - other = merge(other, run_filter(lua, solutions, role="solutions")) + other = merge(other, run_filter(lua, pandoc_reads(solutions, cache_dir, settings), role="solutions")) except (subprocess.CalledProcessError, json.JSONDecodeError) as problem: stderr = getattr(problem, "stderr", None) error = (stderr.decode("utf-8", "replace") if stderr else str(problem)).strip() @@ -693,13 +715,27 @@ def brief(b: dict, depth: int = 0) -> list[str]: return "\n".join(l for b in ast["blocks"] for l in brief(b)) -def write_filter(document: Path, solutions: Optional[Path], backend: Backend) -> tuple[str, Reply]: +def write_filter( + document: Path, + solutions: Optional[Path], + backend: Backend, + *, + cache_dir: Path = Path(".in2lambda-agent"), + settings: Optional[Settings] = None, +) -> tuple[str, Reply]: """Route B's one call: a Lua filter for the structure of this document's set. Where a set writes its solutions in a second document, one filter reads both: the call is shown the structure of each, and the filter it writes tells them apart by the role `run_filter` passes. + + A PDF is shown as the markdown its OCR made (`pandoc_reads`), which is the tree + `convert` then runs the filter over; `cache_dir` and `settings` are where that OCR is + kept and the credentials that fetch it. """ + settings = settings or load_settings() + document = pandoc_reads(document, cache_dir, settings) + solutions = None if solutions is None else pandoc_reads(solutions, cache_dir, settings) version = subprocess.check_output(["pandoc", "--version"]).decode().split()[1] both = "" if solutions is None else f""" @@ -774,7 +810,9 @@ def convert_folder( "not end in `solutions` or `sol`. Name a single document to convert that " "document on its own." ) - lua_source, usage = write_filter(pairs[0][0], pairs[0][1], backend) + lua_source, usage = write_filter( + pairs[0][0], pairs[0][1], backend, cache_dir=cache_dir, settings=settings + ) out_dir = Path(out_dir) out_dir.mkdir(parents=True, exist_ok=True) lua = out_dir / "filter.lua" diff --git a/in2lambda_agent/sweep.py b/in2lambda_agent/sweep.py index 7870b69..54c6056 100644 --- a/in2lambda_agent/sweep.py +++ b/in2lambda_agent/sweep.py @@ -2,11 +2,12 @@ A set is a folder holding at least one questions document. Each set is converted as `routes.convert_folder` converts one: a model call writes a Lua -filter from the first sheet of the set pandoc can read, and each sheet of the -set then runs through route A, the direct model call, and route B, that filter -under pandoc. The sweep runs the sheets itself rather than calling -`convert_folder`, so that it can time each sheet and write a row for a sheet -whose conversion raised. +filter from the first sheet of the set, and each sheet of the set then runs +through route A, the direct model call, and route B, that filter under pandoc. +Pandoc reads a PDF sheet as the markdown its OCR made, so a set of PDFs has a +filter and both routes like any other. The sweep runs the sheets itself rather +than calling `convert_folder`, so that it can time each sheet and write a row +for a sheet whose conversion raised. The measures of docs/plan.md are the columns: how many fields the two routes returned, how many they agreed on, how many the adjudication call decided, how @@ -41,9 +42,6 @@ NO_SET = "no set: " """What the reason of a sheet that built no set begins with.""" -NO_FILTER_PDF = "no filter: pandoc reads no sheet of this set" -"""The reason of a set with no filter because every sheet of it is a PDF.""" - @dataclass class Row: @@ -70,9 +68,8 @@ class Row: reason: Empty where both routes ran and the sheet built its set. `no set: ` where the conversion raised and the sheet built nothing, `route B failed: ` where the set is - route A's alone, and `no filter: ` where no sheet of the set - ran route B, because the filter call did not finish or because - pandoc reads no sheet of the set. + route A's alone, and `no filter: ` where the set's filter call + did not finish and so no sheet of it ran route B. """ set: str @@ -126,16 +123,17 @@ def _sheets( def _filter_pair( pairs: Sequence[tuple[Path, Optional[Path]]] -) -> Optional[tuple[Path, Optional[Path]]]: - """The pair a set's filter is written from: the first pandoc can read. - - `routes.write_filter` shows the call pandoc's tree of the document, and - pandoc cannot read a PDF. A set whose first sheet is a PDF would otherwise - have no filter at all, and lose route B on the sheets pandoc does read. - Where every sheet is a PDF there is nothing to write a filter from, and the - set converts through route A alone. +) -> tuple[Path, Optional[Path]]: + """The pair a set's filter is written from: the first tex, md or docx sheet, + and the first PDF where the set has no other. + + `routes.write_filter` shows the call pandoc's tree of the document, and a + PDF's tree is the tree of the markdown its OCR made, which is a reading of + the printed page rather than the document's own structure. A sheet pandoc + reads itself is the better one to write the filter from, so it is preferred + however far down the set it is. """ - return next((one for one in pairs if one[0].suffix.lower() != ".pdf"), None) + return next((one for one in pairs if one[0].suffix.lower() != ".pdf"), pairs[0]) def sets( @@ -231,9 +229,8 @@ def sweep( """Converts every set of a corpus, printing a line per sheet and writing the table. What a sheet raises is that sheet's row, and the sheets after it still run. - A set with no filter — its call raised, or pandoc reads no sheet of it — - converts every sheet of itself through route A alone, and says so in each - of its rows. + A set whose filter call raised converts every sheet of itself through route + A alone, and says so in each of its rows. Args: root: The corpus directory. @@ -264,19 +261,18 @@ def sweep( lua: Optional[Path] = None no_filter = "" filter_tokens = 0 - readable = _filter_pair(pairs) + written_from = _filter_pair(pairs) started = time.monotonic() - if readable is None: - no_filter = NO_FILTER_PDF + try: + source, usage = routes.write_filter( + written_from[0], written_from[1], backend, cache_dir=cache, settings=settings + ) + except Exception as problem: + no_filter = f"no filter: {_one_line(str(problem))}" else: - try: - source, usage = routes.write_filter(readable[0], readable[1], backend) - except Exception as problem: - no_filter = f"no filter: {_one_line(str(problem))}" - else: - lua = into / "filter.lua" - lua.write_text(source, encoding="utf-8") - filter_tokens = usage.usage.input_tokens + usage.usage.output_tokens + lua = into / "filter.lua" + lua.write_text(source, encoding="utf-8") + filter_tokens = usage.usage.input_tokens + usage.usage.output_tokens filter_seconds = time.monotonic() - started for number, (sheet, solutions) in enumerate(pairs): diff --git a/in2lambda_agent/targets.py b/in2lambda_agent/targets.py index 719e3d9..6cf46b7 100644 --- a/in2lambda_agent/targets.py +++ b/in2lambda_agent/targets.py @@ -325,9 +325,9 @@ def run_one( saved = Path(filters) / target.name reply = saved / REPLY_NAME boxes = saved / AREAS_NAME - # Pandoc reads neither a PDF nor the markdown an OCR made of one back into - # the document's structure, so route B cannot run over a scanned target: - # it converts through route A alone, and no filter is written for it. + # A PDF target converts through route A alone and has no filter, though + # route B now reads a PDF as the markdown its OCR made: a filter here is + # saved under `--filters` and replayed, which is its own ticket. lua = None if target.questions.suffix.lower() == ".pdf" else saved / FILTER_NAME if replay: absent = [one for one in (reply, boxes, lua) if one is not None and not one.is_file()] @@ -356,7 +356,13 @@ def read_back(file: Path): if lua is not None and not lua.is_file(): saved.mkdir(parents=True, exist_ok=True) lua.write_text( - routes.write_filter(target.questions, target.solutions, backend)[0], + routes.write_filter( + target.questions, + target.solutions, + backend, + cache_dir=cache_dir, + settings=settings, + )[0], encoding="utf-8", ) converted = routes.convert( diff --git a/in2lambda_agent/ui/server.py b/in2lambda_agent/ui/server.py index e5d1b1e..58f5cc9 100644 --- a/in2lambda_agent/ui/server.py +++ b/in2lambda_agent/ui/server.py @@ -166,7 +166,13 @@ def _run(self, options: Options) -> None: options.out_dir.mkdir(parents=True, exist_ok=True) lua = options.out_dir / "filter.lua" lua.write_text( - routes.write_filter(options.source, solutions, backend)[0], + routes.write_filter( + options.source, + solutions, + backend, + cache_dir=self.cache_dir, + settings=self.settings, + )[0], encoding="utf-8", ) self._stage("filter", str(lua)) diff --git a/tests/conftest.py b/tests/conftest.py index e676aef..6fe2388 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -4,13 +4,31 @@ import pytest +from in2lambda_agent import ocr from in2lambda_agent.mathpix import MathpixError from in2lambda_agent.model import Reply, ToolCall, Usage +from in2lambda_agent.settings import Settings # A real PNG rather than a few bytes named like one: the set checks compile the # set as the PDF generator does, and xelatex refuses a file it cannot load. PNG = (Path(__file__).parent / "fixtures" / "ball.png").read_bytes() +CREDENTIALS = Settings(mathpix_app_id="id", mathpix_api_key="key") +"""Enough to convert a PDF the cache holds: the conversion builds the client +before it asks the cache, and a client is refused without them.""" + + +def cached_pdf(folder: Path, name: str, cache: Path, markdown: Path) -> Path: + """A PDF sheet whose OCR the cache already holds, so Mathpix is not called.""" + pdf = folder / name + pdf.write_bytes(b"%PDF-1.4 " + name.encode()) + entry = cache / ocr._hash(pdf) + (entry / ocr.MEDIA_NAME).mkdir(parents=True) + (entry / ocr.SOURCE_NAME).write_text( + markdown.read_text(encoding="utf-8"), encoding="utf-8" + ) + return pdf + class FakeBackend: """A model backend that answers from a list instead of calling a model. diff --git a/tests/test_cli.py b/tests/test_cli.py index 39e1beb..d84a984 100644 --- a/tests/test_cli.py +++ b/tests/test_cli.py @@ -266,9 +266,13 @@ def test_the_written_filter_is_kept_in_the_out_directory( ): given = {} monkeypatch.setattr(routes, "convert", records(given, converted(tmp_path / "s.zip"))) - monkeypatch.setattr( - routes, "write_filter", lambda document, solutions, backend: ("-- lua", None) - ) + wrote = {} + + def write_filter(document, solutions, backend, **passed): + wrote.update(passed) + return "-- lua", None + + monkeypatch.setattr(routes, "write_filter", write_filter) code = main( [ @@ -277,12 +281,18 @@ def test_the_written_filter_is_kept_in_the_out_directory( "--write-filter", "--out", str(tmp_path / "out"), + "--cache", + str(tmp_path / "cache"), ] ) assert code == 0 assert (tmp_path / "out" / "filter.lua").read_text() == "-- lua" assert given["lua"] == tmp_path / "out" / "filter.lua" + # The filter call reads a PDF through the OCR, so it is given the run's own + # cache and the settings the conversion beside it was given. + assert wrote["cache_dir"] == tmp_path / "cache" + assert wrote["settings"] is given["settings"] def test_a_flagged_field_does_not_stop_the_build(tmp_path, backend, monkeypatch, capsys): diff --git a/tests/test_routes.py b/tests/test_routes.py index 1296cf9..58b9324 100644 --- a/tests/test_routes.py +++ b/tests/test_routes.py @@ -19,10 +19,10 @@ import pytest -from conftest import FakeBackend +from conftest import CREDENTIALS, FakeBackend, cached_pdf import in2lambda_agent.routes as routes -from in2lambda_agent import cli, pair +from in2lambda_agent import cli, ocr, pair from in2lambda_agent.settings import Settings ME2 = Path(__file__).parent / "fixtures" / "me2" @@ -381,6 +381,51 @@ def test_the_filter_call_sees_both_documents_and_how_to_tell_them_apart(): assert "Energy" not in prompt and "in2lambda_role" not in prompt +def test_pandoc_reads_a_pdf_as_the_markdown_its_ocr_made(tmp_path): + # Pandoc reads no PDF, so route B reads the markdown route A reads. + cache = tmp_path / "cache" + pdf = cached_pdf(tmp_path, "sheet.pdf", cache, FIXTURES / "sheet.md") + + assert routes.pandoc_reads(pdf, cache, CREDENTIALS) == ( + cache / ocr._hash(pdf) / ocr.SOURCE_NAME + ) + # Everything else pandoc reads itself, and no credential is asked for. + for document in (FIXTURES / "sheet.md", FIXTURES / "tex-sheet.tex"): + assert routes.pandoc_reads(document, cache, Settings()) == document + + +@pytest.mark.skipif(shutil.which("pandoc") is None, reason="pandoc") +def test_the_filter_call_for_a_pdf_is_shown_the_tree_of_its_ocr_markdown(tmp_path): + cache = tmp_path / "cache" + pdf = cached_pdf(tmp_path, "sheet.pdf", cache, FIXTURES / "sheet.md") + backend = FakeBackend("function Pandoc(doc) end") + + lua, _ = routes.write_filter(pdf, None, backend, cache_dir=cache, settings=CREDENTIALS) + + ((_, prompt),) = backend.calls + assert lua == "function Pandoc(doc) end" + assert "Header(2): Question 1" in prompt and "Header(2): Solutions" in prompt + + +@pytest.mark.skipif(shutil.which("pandoc") is None, reason="pandoc") +def test_route_b_runs_over_a_pdfs_ocr_markdown_and_agrees_with_route_a(tmp_path): + cache = tmp_path / "cache" + pdf = cached_pdf(tmp_path, "sheet.pdf", cache, FIXTURES / "sheet.md") + + result = routes.convert( + pdf, + out_dir=tmp_path / "out", + cache_dir=cache, + backend=FakeBackend(json.dumps(SHEET_DIRECT)), + settings=CREDENTIALS, + lua=FIXTURES / "pair-filter.lua", + name="sheet", + ) + + assert result.route_b_error is None + assert (result.fields, result.agreed) == (10, 10) + + @pytest.mark.skipif(not PHYS.is_dir() or shutil.which("pandoc") is None, reason="private corpus and pandoc") def test_a_filter_written_for_the_set_reads_a_sheet_with_pandoc_alone(): reply = routes.run_filter(FILTER, PHYS / "mechanics_23-24_PS1.tex") diff --git a/tests/test_sweep.py b/tests/test_sweep.py index a81cf26..1706837 100644 --- a/tests/test_sweep.py +++ b/tests/test_sweep.py @@ -14,10 +14,10 @@ import pytest -from conftest import FakeBackend +from conftest import CREDENTIALS, FakeBackend, cached_pdf from test_routes import PAIRED_DIRECT, SHEET_DIRECT -from in2lambda_agent import cli, ocr, routes, sweep +from in2lambda_agent import cli, routes, sweep from in2lambda_agent.model import ModelError from in2lambda_agent.settings import Settings @@ -37,10 +37,6 @@ ) pandoc = pytest.mark.skipif(shutil.which("pandoc") is None, reason="pandoc") -CREDENTIALS = Settings(mathpix_app_id="id", mathpix_api_key="key") -"""Enough to convert a PDF the cache holds: `markdown_of` builds the client -before it asks the cache, and a client is refused without them.""" - def corpus(root: Path, *names: str) -> Path: """Writes one set folder per name, each holding the three fixture sheets.""" @@ -52,18 +48,6 @@ def corpus(root: Path, *names: str) -> Path: return root -def cached_pdf(folder: Path, name: str, cache: Path, markdown: Path) -> Path: - """A PDF sheet whose OCR the cache already holds, so Mathpix is not called.""" - pdf = folder / name - pdf.write_bytes(b"%PDF-1.4 " + name.encode()) - entry = cache / ocr._hash(pdf) - (entry / ocr.MEDIA_NAME).mkdir(parents=True) - (entry / ocr.SOURCE_NAME).write_text( - markdown.read_text(encoding="utf-8"), encoding="utf-8" - ) - return pdf - - def replies(sets: int = 1) -> list[str]: """The model's answers for one set: the filter, and each sheet's call.""" return [FILTER, json.dumps(PAIRED_DIRECT), ADJUDICATION, json.dumps(SHEET_DIRECT)] * sets @@ -267,8 +251,9 @@ def test_a_set_whose_filter_call_fails_converts_every_sheet_through_route_a(tmp_ @pandoc def test_the_filter_is_written_from_the_first_sheet_pandoc_can_read(tmp_path): - # pandoc cannot read a PDF, and one PDF at the head of a set would otherwise - # deny route B to every sheet of the set, the ones pandoc reads among them. + # A PDF's tree is its OCR's, which is a reading of the printed page; where + # the set has a sheet pandoc reads itself, the filter is written from that + # sheet's own tree instead. root = tmp_path / "corpus" folder = root / "alpha" folder.mkdir(parents=True) @@ -292,17 +277,18 @@ def test_the_filter_is_written_from_the_first_sheet_pandoc_can_read(tmp_path): # The set's first call is the filter, and it was shown paired.md's blocks. assert "Tutorial Sheet 3" in backend.calls[0][1] assert [row.sheet for row in rows] == ["alpha/a_sheet.pdf", "alpha/paired.md"] - # The PDF fails route B on its own; the sheet pandoc reads has its filter. - assert rows[0].built and rows[0].reason.startswith("route B failed: ") + # Both sheets run both routes, the PDF over the markdown its OCR made. + assert rows[0].built and rows[0].reason == "" and rows[0].agreed == 10 assert rows[1].reason == "" and (rows[1].agreed, rows[1].adjudicated) == (8, 1) -def test_a_set_of_pdfs_alone_makes_no_filter_call_and_runs_route_a(tmp_path): +@pandoc +def test_a_set_of_pdfs_alone_writes_its_filter_from_the_ocr_markdown(tmp_path): root = tmp_path / "corpus" folder = root / "alpha" folder.mkdir(parents=True) cached_pdf(folder, "only.pdf", tmp_path / "cache", FIXTURES / "sheet.md") - backend = FakeBackend(json.dumps(SHEET_DIRECT)) + backend = FakeBackend(FILTER, json.dumps(SHEET_DIRECT)) rows = sweep.sweep( root, @@ -314,10 +300,13 @@ def test_a_set_of_pdfs_alone_makes_no_filter_call_and_runs_route_a(tmp_path): backend=backend, ) - # The sheet's direct call and no filter call: pandoc has nothing to read. - assert len(backend.calls) == 1 + # The set's filter call and the sheet's direct call, and the filter call was + # shown the tree of the markdown the OCR made of the PDF. + assert len(backend.calls) == 2 + assert "Header(2): Question 1" in backend.calls[0][1] + assert (tmp_path / "work" / "alpha" / "filter.lua").read_text() == FILTER.strip() assert rows[0].built and rows[0].fields == 10 - assert rows[0].reason == sweep.NO_FILTER_PDF + assert rows[0].reason == "" and rows[0].agreed == 10 @pandoc @@ -368,3 +357,30 @@ def test_the_three_course_folders_sweep_from_the_command_line(tmp_path, capsys): assert header == list(sweep.COLUMNS) assert {one["set"].split("/")[0] for one in written} == set(COURSES) assert code in (0, 1) + + +@live +@pytest.mark.skipif(not (EXAMPLES / "UCL_MechEng").is_dir(), reason="private corpus") +def test_the_ucl_mecheng_pdfs_run_both_routes(tmp_path, capsys): + # The ticket's run: a set whose every sheet is a PDF, which route B now + # reads as the markdown Mathpix made of it. + results = tmp_path / "results.csv" + + code = cli.main( + [ + "corpus", str(EXAMPLES), "UCL_MechEng", + "--suffix", "pdf", + "--results", str(results), + "--work", str(tmp_path / "work"), + "--cache", str(Path.home() / ".cache" / "in2lambda-agent"), + ] + ) + + printed = capsys.readouterr().out + print("\n" + printed) + print(results.read_text(encoding="utf-8")) + _, written = table(results) + assert len(written) == 2 + assert [one["reason"] for one in written] == ["", ""] + assert all(int(one["agreed"]) > 0 for one in written) + assert code == 0 diff --git a/tests/test_targets.py b/tests/test_targets.py index 6f7baf6..0cf6dc2 100644 --- a/tests/test_targets.py +++ b/tests/test_targets.py @@ -14,7 +14,7 @@ import pytest -from conftest import FakeBackend +from conftest import CREDENTIALS, FakeBackend from in2lambda_agent import gate, routes, targets @@ -311,6 +311,29 @@ def test_the_filter_is_written_once_and_read_after_that(tmp_path, monkeypatch): assert calls[1]["lua"] == lua +def test_the_filter_call_is_given_the_runs_cache_and_settings(tmp_path, monkeypatch): + # A target pairs by role and not by suffix, so a tex sheet's solutions may be + # a PDF, which the filter call reads through the OCR. That OCR belongs in the + # run's own cache, and the run's own credentials fetch it. + fake_convert(monkeypatch) + wrote = {} + + def write_filter(document, solutions, backend, **passed): + wrote.update(passed) + return "-- filter", None + + monkeypatch.setattr(targets.routes, "write_filter", write_filter) + make_target(tmp_path / "corpus", "ME2") + (target,) = targets.find(tmp_path / "corpus") + + targets.run_one( + target, filters=tmp_path / "filters", out_dir=tmp_path / "out", + cache_dir=tmp_path / "cache", settings=CREDENTIALS, backend=FakeBackend(), + ) + + assert wrote == {"cache_dir": tmp_path / "cache", "settings": CREDENTIALS} + + def test_a_scanned_target_converts_with_no_filter(tmp_path, monkeypatch): # Pandoc cannot read a PDF, so there is no structure to write a filter from # and no call to make: route A converts the pages' markdown alone. diff --git a/tests/test_ui.py b/tests/test_ui.py index b20a4d6..bfe0168 100644 --- a/tests/test_ui.py +++ b/tests/test_ui.py @@ -313,11 +313,13 @@ def test_a_filter_file_reaches_the_conversion(client, root, tmp_path, monkeypatc def test_the_page_writes_a_filter_and_links_it(client, root, tmp_path, monkeypatch): - monkeypatch.setattr( - server.routes, - "write_filter", - lambda document, solutions, backend: ("function Pandoc(doc) end\n", None), - ) + wrote = {} + + def write_filter(document, solutions, backend, **passed): + wrote.update(passed) + return "function Pandoc(doc) end\n", None + + monkeypatch.setattr(server.routes, "write_filter", write_filter) seen = faked(monkeypatch, zip_path=written(tmp_path / "out", "set.zip")) client.post( @@ -336,6 +338,9 @@ def test_the_page_writes_a_filter_and_links_it(client, root, tmp_path, monkeypat assert seen[0]["lua"] == lua assert [one["name"] for one in found if one["type"] == "stage"] == ["filter"] assert client.get(links["filter.lua"]).text == "function Pandoc(doc) end\n" + # The filter call reads a PDF through the OCR, so it is given the page's own + # cache and settings rather than a default cache under the working directory. + assert wrote == {"cache_dir": tmp_path / "cache", "settings": Settings()} def test_a_stage_reaches_the_page_before_the_run_ends(client, app, root, monkeypatch):