Skip to content
20 changes: 10 additions & 10 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -197,8 +197,8 @@ pdf` runs the PDFs too, which needs Mathpix credentials and one call per PDF.

A set is a folder holding at least one questions document. Each set converts as a
folder run of `convert` does: one model call writes the set's filter from the first
sheet of it pandoc can read, and each sheet of the set then runs through route A and
that filter. A sheet and the solutions file beside it are one conversion and one row,
sheet of it pandoc reads itself, or from the first PDF where the set has no other, and
each sheet of the set then runs through route A and that filter. A sheet and the solutions file beside it are one conversion and one row,
named after the questions file. A folder of figures, a tex drawing with no
`\begin{document}` among them, and a folder holding a solutions file alone, are not
sets and have no row.
Expand All @@ -217,11 +217,12 @@ tokens, seconds, reason

`reason` is empty where the sheet ran through both routes
and built its set. One sheet whose conversion raises is one row, with `no set:` and the
error as its reason, and the sheets after it still run. A set with no filter — its
filter call did not finish, or every sheet of it is a PDF, which pandoc cannot read —
converts every sheet through route A alone, and each of those rows reads `no filter:`
and why. A PDF among sheets pandoc reads leaves the rest of the set on both routes: the
filter is written from one of those sheets, and the PDF alone fails route B.
error as its reason, and the sheets after it still run. A set whose filter call did not
finish converts every sheet through route A alone, and each of those rows reads `no
filter:` and why. A PDF runs route B over the markdown its OCR made, which is the
markdown route A reads, so a set of PDFs has a filter like any other; where the filter
was written from a tex or docx sheet beside it, that filter may still fail on the PDF,
and that row alone reads `route B failed:`.

`corpus` exits 0 where it ran at least one sheet and every sheet built a set. It exits 1
where it found no sheet, and where any sheet built no set — a row whose reason begins
Expand Down Expand Up @@ -311,9 +312,8 @@ Those are the only calls a saved target spares. A target with a filter runs rout
on every run, and a model adjudicates every field the two routes word differently. A
verdict can go the other way on a later run, so the wording of a difference and the
`flagged` count move from run to run while the fields `differs.txt` accepts stay
accepted. A target whose questions document is a PDF has no filter:
pandoc cannot read a PDF, so route B does not run and route A converts the pages'
markdown alone. `--out` (default `./out`) is where each target's set is written, under
accepted. A target whose questions document is a PDF has no filter: route B does not
run, and route A converts the pages' markdown alone. `--out` (default `./out`) is where each target's set is written, under
the target's own name, and `--cache` (default `./.in2lambda-agent`) is where the OCR of
each PDF is kept.

Expand Down
5 changes: 4 additions & 1 deletion in2lambda_agent/cli.py
Original file line number Diff line number Diff line change
Expand Up @@ -277,7 +277,10 @@ def convert_command(args: argparse.Namespace) -> int:
out_dir.mkdir(parents=True, exist_ok=True)
lua = out_dir / "filter.lua"
lua.write_text(
routes.write_filter(document, solutions, backend)[0], encoding="utf-8"
routes.write_filter(
document, solutions, backend, cache_dir=args.cache, settings=settings
)[0],
encoding="utf-8",
)
result = routes.convert(
document,
Expand Down
56 changes: 47 additions & 9 deletions in2lambda_agent/routes.py
Original file line number Diff line number Diff line change
Expand Up @@ -489,14 +489,33 @@ def _areas_counted(areas: dict[str, list[dict[str, Any]]]) -> str:
_UNDERLINE = Path(__file__).parent / "underline.lua"


def _ocr(document: Path, cache_dir: Path, settings: Settings):
"""The OCR of a PDF, from the cache where it holds one."""
from in2lambda_agent.mathpix import MathpixClient
from in2lambda_agent.ocr import ocr_pdf

return ocr_pdf(document, cache_dir=cache_dir, client=MathpixClient.from_settings(settings))


def pandoc_reads(document: Path, cache_dir: Path, settings: Settings) -> Path:
"""The file pandoc is given for a document: for a PDF, its OCR markdown.

Pandoc reads no PDF, so both halves of route B - the tree the filter is
written from, and the run of that filter - read the markdown Mathpix made of
the pages, which is what route A reads too. Every other document pandoc
reads itself, and no credential is asked for.
"""
document = Path(document)
if document.suffix.lower() == ".pdf":
return _ocr(document, cache_dir, settings).markdown
return document


def markdown_of(document: Path, cache_dir: Path, settings: Settings) -> tuple[str, Path]:
"""The document as markdown, and the folder its images are in."""
document = Path(document)
if document.suffix.lower() == ".pdf":
from in2lambda_agent.mathpix import MathpixClient
from in2lambda_agent.ocr import ocr_pdf

ocr = ocr_pdf(document, cache_dir=cache_dir, client=MathpixClient.from_settings(settings))
ocr = _ocr(document, cache_dir, settings)
return ocr.markdown.read_text(encoding="utf-8"), ocr.markdown.parent
if document.suffix.lower() in (".md", ".markdown"):
return document.read_text(encoding="utf-8"), document.parent
Expand Down Expand Up @@ -545,7 +564,8 @@ def convert(
"""Route A, route B where a filter is given, reconcile, verify, write.

Route B reads the solutions document too, under its own role, and the two runs are
merged before the comparison. Where a filter run fails, the route A reply is the
merged before the comparison. Of a PDF it reads the markdown the OCR made, which is
what route A reads. Where a filter run fails, the route A reply is the
result and `route_b_error` holds pandoc's message, so that one sheet of a folder does
not stop the other eight.

Expand Down Expand Up @@ -601,9 +621,11 @@ def said(stage: str, message: str) -> None:
said("route B", "did not run: no filter")
else:
try:
other = run_filter(lua, document)
# A PDF's filter runs over the markdown its OCR made, which
# `markdown_of` has by now put in the cache.
other = run_filter(lua, pandoc_reads(document, cache_dir, settings))
if solutions is not None:
other = merge(other, run_filter(lua, solutions, role="solutions"))
other = merge(other, run_filter(lua, pandoc_reads(solutions, cache_dir, settings), role="solutions"))
except (subprocess.CalledProcessError, json.JSONDecodeError) as problem:
stderr = getattr(problem, "stderr", None)
error = (stderr.decode("utf-8", "replace") if stderr else str(problem)).strip()
Expand Down Expand Up @@ -693,13 +715,27 @@ def brief(b: dict, depth: int = 0) -> list[str]:
return "\n".join(l for b in ast["blocks"] for l in brief(b))


def write_filter(document: Path, solutions: Optional[Path], backend: Backend) -> tuple[str, Reply]:
def write_filter(
document: Path,
solutions: Optional[Path],
backend: Backend,
*,
cache_dir: Path = Path(".in2lambda-agent"),
settings: Optional[Settings] = None,
) -> tuple[str, Reply]:
"""Route B's one call: a Lua filter for the structure of this document's set.

Where a set writes its solutions in a second document, one filter reads both: the
call is shown the structure of each, and the filter it writes tells them apart by the
role `run_filter` passes.

A PDF is shown as the markdown its OCR made (`pandoc_reads`), which is the tree
`convert` then runs the filter over; `cache_dir` and `settings` are where that OCR is
kept and the credentials that fetch it.
"""
settings = settings or load_settings()
document = pandoc_reads(document, cache_dir, settings)
solutions = None if solutions is None else pandoc_reads(solutions, cache_dir, settings)
version = subprocess.check_output(["pandoc", "--version"]).decode().split()[1]
both = "" if solutions is None else f"""

Expand Down Expand Up @@ -774,7 +810,9 @@ def convert_folder(
"not end in `solutions` or `sol`. Name a single document to convert that "
"document on its own."
)
lua_source, usage = write_filter(pairs[0][0], pairs[0][1], backend)
lua_source, usage = write_filter(
pairs[0][0], pairs[0][1], backend, cache_dir=cache_dir, settings=settings
)
out_dir = Path(out_dir)
out_dir.mkdir(parents=True, exist_ok=True)
lua = out_dir / "filter.lua"
Expand Down
64 changes: 30 additions & 34 deletions in2lambda_agent/sweep.py
Original file line number Diff line number Diff line change
Expand Up @@ -2,11 +2,12 @@

A set is a folder holding at least one questions document. Each set is
converted as `routes.convert_folder` converts one: a model call writes a Lua
filter from the first sheet of the set pandoc can read, and each sheet of the
set then runs through route A, the direct model call, and route B, that filter
under pandoc. The sweep runs the sheets itself rather than calling
`convert_folder`, so that it can time each sheet and write a row for a sheet
whose conversion raised.
filter from the first sheet of the set, and each sheet of the set then runs
through route A, the direct model call, and route B, that filter under pandoc.
Pandoc reads a PDF sheet as the markdown its OCR made, so a set of PDFs has a
filter and both routes like any other. The sweep runs the sheets itself rather
than calling `convert_folder`, so that it can time each sheet and write a row
for a sheet whose conversion raised.

The measures of docs/plan.md are the columns: how many fields the two routes
returned, how many they agreed on, how many the adjudication call decided, how
Expand Down Expand Up @@ -41,9 +42,6 @@
NO_SET = "no set: "
"""What the reason of a sheet that built no set begins with."""

NO_FILTER_PDF = "no filter: pandoc reads no sheet of this set"
"""The reason of a set with no filter because every sheet of it is a PDF."""


@dataclass
class Row:
Expand All @@ -70,9 +68,8 @@ class Row:
reason: Empty where both routes ran and the sheet built its set.
`no set: <error>` where the conversion raised and the sheet built
nothing, `route B failed: <pandoc's message>` where the set is
route A's alone, and `no filter: <why>` where no sheet of the set
ran route B, because the filter call did not finish or because
pandoc reads no sheet of the set.
route A's alone, and `no filter: <why>` where the set's filter call
did not finish and so no sheet of it ran route B.
"""

set: str
Expand Down Expand Up @@ -126,16 +123,17 @@ def _sheets(

def _filter_pair(
pairs: Sequence[tuple[Path, Optional[Path]]]
) -> Optional[tuple[Path, Optional[Path]]]:
"""The pair a set's filter is written from: the first pandoc can read.

`routes.write_filter` shows the call pandoc's tree of the document, and
pandoc cannot read a PDF. A set whose first sheet is a PDF would otherwise
have no filter at all, and lose route B on the sheets pandoc does read.
Where every sheet is a PDF there is nothing to write a filter from, and the
set converts through route A alone.
) -> tuple[Path, Optional[Path]]:
"""The pair a set's filter is written from: the first tex, md or docx sheet,
and the first PDF where the set has no other.

`routes.write_filter` shows the call pandoc's tree of the document, and a
PDF's tree is the tree of the markdown its OCR made, which is a reading of
the printed page rather than the document's own structure. A sheet pandoc
reads itself is the better one to write the filter from, so it is preferred
however far down the set it is.
"""
return next((one for one in pairs if one[0].suffix.lower() != ".pdf"), None)
return next((one for one in pairs if one[0].suffix.lower() != ".pdf"), pairs[0])


def sets(
Expand Down Expand Up @@ -231,9 +229,8 @@ def sweep(
"""Converts every set of a corpus, printing a line per sheet and writing the table.

What a sheet raises is that sheet's row, and the sheets after it still run.
A set with no filter — its call raised, or pandoc reads no sheet of it —
converts every sheet of itself through route A alone, and says so in each
of its rows.
A set whose filter call raised converts every sheet of itself through route
A alone, and says so in each of its rows.

Args:
root: The corpus directory.
Expand Down Expand Up @@ -264,19 +261,18 @@ def sweep(
lua: Optional[Path] = None
no_filter = ""
filter_tokens = 0
readable = _filter_pair(pairs)
written_from = _filter_pair(pairs)
started = time.monotonic()
if readable is None:
no_filter = NO_FILTER_PDF
try:
source, usage = routes.write_filter(
written_from[0], written_from[1], backend, cache_dir=cache, settings=settings
)
except Exception as problem:
no_filter = f"no filter: {_one_line(str(problem))}"
else:
try:
source, usage = routes.write_filter(readable[0], readable[1], backend)
except Exception as problem:
no_filter = f"no filter: {_one_line(str(problem))}"
else:
lua = into / "filter.lua"
lua.write_text(source, encoding="utf-8")
filter_tokens = usage.usage.input_tokens + usage.usage.output_tokens
lua = into / "filter.lua"
lua.write_text(source, encoding="utf-8")
filter_tokens = usage.usage.input_tokens + usage.usage.output_tokens
filter_seconds = time.monotonic() - started

for number, (sheet, solutions) in enumerate(pairs):
Expand Down
14 changes: 10 additions & 4 deletions in2lambda_agent/targets.py
Original file line number Diff line number Diff line change
Expand Up @@ -325,9 +325,9 @@ def run_one(
saved = Path(filters) / target.name
reply = saved / REPLY_NAME
boxes = saved / AREAS_NAME
# Pandoc reads neither a PDF nor the markdown an OCR made of one back into
# the document's structure, so route B cannot run over a scanned target:
# it converts through route A alone, and no filter is written for it.
# A PDF target converts through route A alone and has no filter, though
# route B now reads a PDF as the markdown its OCR made: a filter here is
# saved under `--filters` and replayed, which is its own ticket.
lua = None if target.questions.suffix.lower() == ".pdf" else saved / FILTER_NAME
if replay:
absent = [one for one in (reply, boxes, lua) if one is not None and not one.is_file()]
Expand Down Expand Up @@ -356,7 +356,13 @@ def read_back(file: Path):
if lua is not None and not lua.is_file():
saved.mkdir(parents=True, exist_ok=True)
lua.write_text(
routes.write_filter(target.questions, target.solutions, backend)[0],
routes.write_filter(
target.questions,
target.solutions,
backend,
cache_dir=cache_dir,
settings=settings,
)[0],
encoding="utf-8",
)
converted = routes.convert(
Expand Down
8 changes: 7 additions & 1 deletion in2lambda_agent/ui/server.py
Original file line number Diff line number Diff line change
Expand Up @@ -166,7 +166,13 @@ def _run(self, options: Options) -> None:
options.out_dir.mkdir(parents=True, exist_ok=True)
lua = options.out_dir / "filter.lua"
lua.write_text(
routes.write_filter(options.source, solutions, backend)[0],
routes.write_filter(
options.source,
solutions,
backend,
cache_dir=self.cache_dir,
settings=self.settings,
)[0],
encoding="utf-8",
)
self._stage("filter", str(lua))
Expand Down
18 changes: 18 additions & 0 deletions tests/conftest.py
Original file line number Diff line number Diff line change
Expand Up @@ -4,13 +4,31 @@

import pytest

from in2lambda_agent import ocr
from in2lambda_agent.mathpix import MathpixError
from in2lambda_agent.model import Reply, ToolCall, Usage
from in2lambda_agent.settings import Settings

# A real PNG rather than a few bytes named like one: the set checks compile the
# set as the PDF generator does, and xelatex refuses a file it cannot load.
PNG = (Path(__file__).parent / "fixtures" / "ball.png").read_bytes()

CREDENTIALS = Settings(mathpix_app_id="id", mathpix_api_key="key")
"""Enough to convert a PDF the cache holds: the conversion builds the client
before it asks the cache, and a client is refused without them."""


def cached_pdf(folder: Path, name: str, cache: Path, markdown: Path) -> Path:
"""A PDF sheet whose OCR the cache already holds, so Mathpix is not called."""
pdf = folder / name
pdf.write_bytes(b"%PDF-1.4 " + name.encode())
entry = cache / ocr._hash(pdf)
(entry / ocr.MEDIA_NAME).mkdir(parents=True)
(entry / ocr.SOURCE_NAME).write_text(
markdown.read_text(encoding="utf-8"), encoding="utf-8"
)
return pdf


class FakeBackend:
"""A model backend that answers from a list instead of calling a model.
Expand Down
16 changes: 13 additions & 3 deletions tests/test_cli.py
Original file line number Diff line number Diff line change
Expand Up @@ -266,9 +266,13 @@ def test_the_written_filter_is_kept_in_the_out_directory(
):
given = {}
monkeypatch.setattr(routes, "convert", records(given, converted(tmp_path / "s.zip")))
monkeypatch.setattr(
routes, "write_filter", lambda document, solutions, backend: ("-- lua", None)
)
wrote = {}

def write_filter(document, solutions, backend, **passed):
wrote.update(passed)
return "-- lua", None

monkeypatch.setattr(routes, "write_filter", write_filter)

code = main(
[
Expand All @@ -277,12 +281,18 @@ def test_the_written_filter_is_kept_in_the_out_directory(
"--write-filter",
"--out",
str(tmp_path / "out"),
"--cache",
str(tmp_path / "cache"),
]
)

assert code == 0
assert (tmp_path / "out" / "filter.lua").read_text() == "-- lua"
assert given["lua"] == tmp_path / "out" / "filter.lua"
# The filter call reads a PDF through the OCR, so it is given the run's own
# cache and the settings the conversion beside it was given.
assert wrote["cache_dir"] == tmp_path / "cache"
assert wrote["settings"] is given["settings"]


def test_a_flagged_field_does_not_stop_the_build(tmp_path, backend, monkeypatch, capsys):
Expand Down
Loading
Loading