diff --git a/.github/workflows/test.yml b/.github/workflows/test.yml index 8f58b51..033c54c 100644 --- a/.github/workflows/test.yml +++ b/.github/workflows/test.yml @@ -30,7 +30,7 @@ jobs: - name: Install package and test dependencies run: | if [ "${{ matrix.package }}" = "document2md" ]; then - pip install -e ".[test]" + pip install -e ".[mineru,test]" else pip install -e "dof2md[test]" fi @@ -42,12 +42,29 @@ jobs: pytest dof2md/tests fi - # Runs the documentation's worked examples for real, separate from the - # pytest matrix above. Read the Docs' own build stays HTML-only. The OCR - # examples are marked `# doctest: +SKIP` (entering a BatchConverter starts a - # real mineru-api server) and are verified instead by tests/test_batch.py - # and tests/test_cli.py — the exception is written down on the page itself. - docs-doctest: + # A light install with no mineru extra and no apt packages: proves the + # suite and the whole import graph need no mineru at all, i.e. that + # `pip install document2md` (no extra) is genuinely usable on its own. + document2md-light: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + - uses: actions/setup-python@v5 + with: + python-version: "3.12" + - name: Install package and test dependencies + run: pip install -e ".[test]" + - name: Run tests + run: pytest tests + + # Sphinx gate for the Read the Docs developer site: a strict HTML build + # (any broken cross-reference or other warning fails it) plus the + # documentation's few real, executed examples, separate from the pytest + # matrix above. Usage examples that would need mineru or pymupdf4llm live + # on the Pages site (website/) instead, not here, so there is little left + # to mark `# doctest: +SKIP` — the exception, where one remains, is written + # down on the page itself. + docs: runs-on: ubuntu-latest steps: - uses: actions/checkout@v4 @@ -59,9 +76,12 @@ jobs: - name: Install docs and package dependencies run: | pip install -r docs/requirements.txt - # --no-deps: keeps the heavy mineru[pipeline] install out of this - # job, which is also why .readthedocs.yaml installs it this way. + # --no-deps: keeps the heavy mineru[pipeline] install (and + # pymupdf4llm) out of this job, which is also why + # .readthedocs.yaml installs it this way. pip install --no-deps -e . + - name: Build the docs strictly + run: python -m sphinx -n -W --keep-going -b html docs/source docs/build/html - name: Run the documentation's doctest examples run: python -m sphinx -b doctest docs/source docs/build/doctest diff --git a/.github/workflows/website.yml b/.github/workflows/website.yml new file mode 100644 index 0000000..1ce8faa --- /dev/null +++ b/.github/workflows/website.yml @@ -0,0 +1,48 @@ +name: Publish website + +on: + pull_request: + paths: + - "website/**" + - ".github/workflows/website.yml" + push: + branches: [main] + paths: + - "website/**" + - ".github/workflows/website.yml" + workflow_dispatch: + +jobs: + render: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + + - uses: quarto-dev/quarto-actions/setup@v2 + with: + version: 1.9.38 + + - run: quarto render website + + publish: + if: github.event_name != 'pull_request' + needs: render + runs-on: ubuntu-latest + permissions: + contents: write + concurrency: + group: website + cancel-in-progress: true + steps: + - uses: actions/checkout@v4 + + - uses: quarto-dev/quarto-actions/setup@v2 + with: + version: 1.9.38 + + - uses: quarto-dev/quarto-actions/publish@v2 + with: + target: gh-pages + path: website + env: + GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }} diff --git a/CLAUDE.md b/CLAUDE.md index b0d699f..55aae2e 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -6,12 +6,19 @@ code in this repository. ## What this is `document2md` converts a PDF, or an ordered set of scanned page images, into -Markdown, via OCR and layout analysis. It is a wrapper around -[mineru](https://github.com/opendatalab/MinerU); its own contribution is: +Markdown, via one of two backends: [mineru](https://github.com/opendatalab/MinerU) +(OCR and layout analysis, for scanned pages) or +[pymupdf4llm](https://github.com/pymupdf/RAG) (reading a born-digital PDF's own +embedded text layer, no OCR needed — see `document2md/pymupdf_backend.py`). +Its own contribution is: - keeping mineru's `mineru-api` server warm across a batch of documents (`BatchConverter`), instead of paying its startup and model-loading cost once per document; +- resolving which backend converts a given document (`BatchConverter`'s + `backend=`/the CLI's `--backend`; see the seam bullet below) and rejecting, + rather than silently OCR-falling-back on, input the resolved backend can't + handle; - stitching the OCR of several page images of the same document into one continuous Markdown document; - rewriting the raw HTML tables mineru falls back to (rowspan/colspan) into @@ -20,6 +27,9 @@ Markdown, via OCR and layout analysis. It is a wrapper around one's title in the OCR'd text (`cutter`) — a scanned page usually holds the tail of one document and the head of the next. +PyMuPDF and pymupdf4llm are AGPL-3.0 licensed; `document2md` itself stays +Apache-2.0 (see the README's Install section). + It was extracted from the [LegalIA](https://github.com/INGEOTEC/LegalIA) monorepo at commit `e1f258c` (issue [#233](https://github.com/INGEOTEC/LegalIA/issues/233)), where its whole @@ -35,11 +45,17 @@ earlier history is still readable — as `packages/document2md`, and as what it does, not after the corpus that first needed it.* - **It downloads nothing.** It only ever converts a PDF or images already on disk. Getting a document is the caller's problem. -- **There is no backend seam yet.** `mineru` is an implementation detail, and a - cloud OCR/layout service is a plausible second backend — that possibility is - the reason for the name, not something implemented here. No `backend=` - parameter, no registry, no second converter. It gets its own issue when it - happens. +- **The backend seam has two backends behind it.** `BatchConverter(backend=...)` + and the CLI's `--backend` accept `auto` (default), `mineru` and `pymupdf`. + `auto` resolves to `mineru` when its CLI is on `PATH` (mineru remains the + reference backend where installed — it isn't itself replaced by `pymupdf`), + otherwise to `pymupdf`, which reads a PDF's own embedded text layer instead + of OCR-ing it. `mineru` is the `document2md[mineru]` extra rather than a + hard dependency; `pymupdf4llm` (the `pymupdf` backend's dependency) is a + core one, small enough that a bare `pip install document2md` still converts + something. Neither backend silently falls back to the other: a list of page + images, or a PDF without enough of an embedded text layer, under `pymupdf` + raises `RuntimeError` naming the `mineru` extra instead. ## Layout @@ -47,14 +63,23 @@ earlier history is still readable — as `packages/document2md`, and as pyproject.toml document2md itself; the repository *is* the package setup.py two-line setuptools shim document2md/ the package: cli, batch, mineru_server, converter, - tables, cutter + pymupdf_backend, tables, cutter tests/ its pytest suite dof2md/ the old PyPI name's tombstone (see below) scripts/check_package_versions.py -docs/ Sphinx site, published at document2md.readthedocs.io -.github/workflows/ test.yml, publish-pypi.yml +website/ Quarto user site, published to GitHub Pages at + ingeotec.github.io/document2md — install, CLI, + Python, backends; no API reference +docs/ Sphinx developer site, published at + document2md.readthedocs.io — architecture, backends, + development, full API reference; no usage guides + beyond a pointer to the Pages site +.github/workflows/ test.yml, website.yml, publish-pypi.yml ``` +The two sites have two audiences and no overlapping page: the Pages site is +for using the package, Read the Docs for extending it. + Two packages are published from this repository, so the release tag convention is `-v` (`document2md-v0.3.0`, `dof2md-v0.3.0`) rather than a bare `v*`, which would not say which one. @@ -86,8 +111,9 @@ bare `v*`, which would not say which one. pytest tests # document2md pytest dof2md/tests # the tombstone python scripts/check_package_versions.py +python -m sphinx -n -W --keep-going -b html docs/source docs/build/html python -m sphinx -b doctest docs/source docs/build/doctest -python -m sphinx -b html docs/source docs/build/html +quarto render website # the Pages user site; needs Quarto 1.9.38, no Python ``` **The two pytest runs are two invocations on purpose, never a bare `pytest`.** @@ -103,17 +129,28 @@ reason. The tests never import `mineru` — they mock the subprocess boundary (`document2md.converter.convert_to_markdown` / `convert_images_to_markdown`, `document2md.batch.MineruServer`) — so a local -install with `pip install --no-deps -e .` plus `requests` and `pytest` runs -the whole suite without the gigabytes of `mineru[pipeline]`. CI installs the -package fully (`pip install -e ".[test]"`, plus `libgl1`/`libglib2.0-0` for -mineru's opencv) on every supported Python, which is what proves the declared -dependency actually installs. - -The docs' OCR examples are the one documented exception to "every public -symbol has a verified example": entering a `BatchConverter` starts a real -`mineru-api` server, so they are marked `# doctest: +SKIP` and verified -instead by `tests/test_batch.py` and `tests/test_cli.py`. The exception is -written on the docs page itself, not silently skipped. +install with `pip install --no-deps -e .` plus `requests`, `pymupdf4llm` and +`pytest` runs the whole suite without the gigabytes of `mineru[pipeline]`. +`tests/test_pymupdf_backend.py` is the one exception: it exercises the real +`pymupdf`/`pymupdf4llm`, building its own PDFs on the fly with PyMuPDF, since +that backend is small and core rather than optional. CI's `document2md` job +installs the package fully (`pip install -e ".[mineru,test]"`, plus +`libgl1`/`libglib2.0-0` for mineru's opencv), proving the declared `mineru` +extra actually installs; `document2md-light` installs `-e ".[test]"` with no +apt step, proving the rest of the suite (including the `pymupdf` backend) +needs no mineru at all. + +Usage examples that would need `mineru` or `pymupdf4llm` (entering a +`BatchConverter`, running the CLI) live on the Pages site (`website/`) now, +not on Read the Docs — that site's code blocks are illustrative and not +executed at all, and say so once per page. Read the Docs itself currently +has no `# doctest: +SKIP` example: `document2md.cutter`'s real, +executed example on `architecture.rst` is the only one on the site, and it +needs neither backend nor any file on disk. If a future page needs one +(a new backend's own example, say), verify it for real instead in the +matching test module (following `tests/test_batch.py`, `tests/test_cli.py` +and `tests/test_pymupdf_backend.py`) and write the exception on the page +itself, not silently. ## Publishing @@ -126,6 +163,13 @@ set on this repository. `publish-pypi.yml` refuses a tag that disagrees with release whose `ocr` extra requires it** (`document2md>=0.3.0`), or `pip install nota2md[ocr]` breaks for everyone outside the LegalIA repository. +The Pages site publishes itself: `website.yml` renders `website/` on every +pull request that touches it, and publishes to the `gh-pages` branch on every +push to `main` that does — no human action once the workflow exists. Read +the Docs is the opposite: the project must be imported once, by a human, at +readthedocs.org; nothing in this repository can do that, and +`document2md.readthedocs.io` answers 404 until it happens. + ## Language policy Everything written into this repository is in English: identifiers, comments, diff --git a/README.md b/README.md index 303457a..6531a73 100644 --- a/README.md +++ b/README.md @@ -5,10 +5,12 @@ Converts a PDF, or a set of scanned page images, into Markdown — any document, such as an edition of Mexico's official gazette (DOF, *Diario -Oficial de la Federación*) — optionally cropped down to a single note. -It's a wrapper -around [mineru](https://github.com/opendatalab/MinerU) for the OCR/layout -analysis itself; `document2md`'s own contribution is: +Oficial de la Federación*) — optionally cropped down to a single note. It +has two backends: [mineru](https://github.com/opendatalab/MinerU) (OCR and +layout analysis, for scanned pages) and +[pymupdf4llm](https://github.com/pymupdf/RAG) (reading a born-digital PDF's +own embedded text layer, no OCR needed); `document2md`'s own contribution +is: - Keeping mineru's `mineru-api` server warm across a batch of documents, instead of paying its startup (and model-loading) cost once per document. @@ -25,16 +27,38 @@ monorepo at commit `e1f258c`, where every commit of its earlier history (as `packages/document2md`, and as `packages/dof2md` before the rename) can still be read. +## Documentation + +- **Using the package:** [ingeotec.github.io/document2md](https://ingeotec.github.io/document2md/) +- **Extending the package:** [document2md.readthedocs.io](https://document2md.readthedocs.io/) + ## Install ```bash pip install document2md ``` +converts born-digital PDFs (`--backend pymupdf`, via +[pymupdf4llm](https://github.com/pymupdf/RAG)) out of the box — no OCR, no +models. Note: PyMuPDF and pymupdf4llm are AGPL-3.0 licensed, unlike the rest +of `document2md` (Apache-2.0); check that fits your project before +redistributing. Scanned documents still need mineru's OCR/layout models — +add the `mineru` backend for that: + +```bash +pip install "document2md[mineru]" +``` + +`--backend`/`BatchConverter(backend=...)` name each backend explicitly so +`auto`'s policy (prefer `mineru` when installed, fall back to `pymupdf` +otherwise) is visible rather than implicit. `nota2md`'s `ocr` extra (in the +LegalIA repository) must depend on `document2md[mineru]>=0.4.0`, not a bare +`document2md>=0.3.0`, or `pip install nota2md[ocr]` stops installing mineru. + For development, from a clone of this repository: ```bash -pip install -e ".[test]" +pip install -e ".[mineru,test]" ``` ## Usage @@ -82,6 +106,14 @@ flags: rendered PDFs...) in `/_mineru/` instead of discarding it; useful when a conversion looks wrong and mineru's own read of the page is the first thing worth inspecting. +- `--backend {auto,mineru,pymupdf}` (default `auto`) — which backend + converts the document. `auto` resolves to `mineru` when it's on `PATH`, + otherwise `pymupdf`. `pymupdf` reads a PDF's own embedded text layer + instead of running OCR — far faster, at the cost of slightly worse + structure — and can't handle scanned page images or a PDF without enough + of a text layer; `document2md` exits with a message telling you to + `pip install "document2md[mineru]"` instead of a traceback when that + happens, or when `mineru` is requested but isn't installed. ### Python: batch conversion @@ -107,7 +139,12 @@ spanning several scanned pages, and writes the result to `outdir/filename`. The same `titulo`/`titulo_siguiente`, `min_confidence`, `keep_pages` and `keep_mineru_output` options the CLI exposes are also its keyword arguments — see `BatchConverter.__call__`'s docstring for the full -signature. +signature. `BatchConverter(backend="auto")` (the default) picks which +backend does the conversion: `"mineru"` when it's on `PATH`, otherwise +`"pymupdf"`; the resolved name is available as `convert.backend` once +entered. `"pymupdf"` raises `RuntimeError` (naming the `mineru` extra) +rather than silently falling back to it on input it can't handle — a list +of page images, or a PDF without enough of an embedded text layer. `nota2md.legal_provisions` accepts an already-`__enter__`'d `BatchConverter` as its own `converter` parameter, so a batch of DOF legal provisions can @@ -116,5 +153,6 @@ share the same warm server too. ## Tests ```bash -pytest -v +pytest tests +pytest dof2md/tests ``` diff --git a/docs/Makefile b/docs/Makefile new file mode 100644 index 0000000..0810dbb --- /dev/null +++ b/docs/Makefile @@ -0,0 +1,20 @@ +# Minimal makefile for Sphinx documentation +# + +# You can set these variables from the command line. +SPHINXOPTS = +SPHINXBUILD = sphinx-build +SPHINXPROJ = document2md +SOURCEDIR = source +BUILDDIR = build + +# Put it first so that "make" without argument is like "make help". +help: + @$(SPHINXBUILD) -M help "$(SOURCEDIR)" "$(BUILDDIR)" $(SPHINXOPTS) $(O) + +.PHONY: help Makefile + +# Catch-all target: route all unknown targets to Sphinx using the new +# "make mode" option. $(O) is meant as a shortcut for $(SPHINXOPTS). +%: Makefile + @$(SPHINXBUILD) -M $@ "$(SOURCEDIR)" "$(BUILDDIR)" $(SPHINXOPTS) $(O) diff --git a/docs/source/api/batch.rst b/docs/source/api/batch.rst new file mode 100644 index 0000000..427cdeb --- /dev/null +++ b/docs/source/api/batch.rst @@ -0,0 +1,10 @@ +``document2md.batch`` +====================== + +:py:class:`~document2md.batch.BatchConverter`, the package's public Python +entry point, and the backend selector it resolves and dispatches on. + +.. automodule:: document2md.batch + :members: + :private-members: + :undoc-members: diff --git a/docs/source/api/cli.rst b/docs/source/api/cli.rst new file mode 100644 index 0000000..b3900a7 --- /dev/null +++ b/docs/source/api/cli.rst @@ -0,0 +1,9 @@ +``document2md.cli`` +==================== + +The ``document2md`` console script's argument parser and entry point. + +.. automodule:: document2md.cli + :members: + :private-members: + :undoc-members: diff --git a/docs/source/api/converter.rst b/docs/source/api/converter.rst new file mode 100644 index 0000000..dc3eded --- /dev/null +++ b/docs/source/api/converter.rst @@ -0,0 +1,10 @@ +``document2md.converter`` +=========================== + +Shells out to mineru to OCR a PDF or a set of scanned page images into +Markdown, then rewrites its raw HTML table fallback. + +.. automodule:: document2md.converter + :members: + :private-members: + :undoc-members: diff --git a/docs/source/api/cutter.rst b/docs/source/api/cutter.rst new file mode 100644 index 0000000..58e8357 --- /dev/null +++ b/docs/source/api/cutter.rst @@ -0,0 +1,11 @@ +``document2md.cutter`` +======================== + +Slices the Markdown converted from a section's scanned page images down to +just the section of interest, by locating its title and the next +section's title. + +.. automodule:: document2md.cutter + :members: + :private-members: + :undoc-members: diff --git a/docs/source/api/index.rst b/docs/source/api/index.rst new file mode 100644 index 0000000..1391df5 --- /dev/null +++ b/docs/source/api/index.rst @@ -0,0 +1,19 @@ +API reference +============= + +One page per module in :py:mod:`document2md`, in pipeline order (see +:doc:`../architecture`). Every class and function is documented, including +private/internal helpers (leading-underscore names) — useful when +extending or debugging the package, though they are not part of its +public API and can change without notice. + +.. toctree:: + :maxdepth: 1 + + cli + batch + mineru_server + converter + pymupdf_backend + tables + cutter diff --git a/docs/source/api/mineru_server.rst b/docs/source/api/mineru_server.rst new file mode 100644 index 0000000..ce30bdf --- /dev/null +++ b/docs/source/api/mineru_server.rst @@ -0,0 +1,9 @@ +``document2md.mineru_server`` +=============================== + +Manages a persistent ``mineru-api`` process for batch conversions. + +.. automodule:: document2md.mineru_server + :members: + :private-members: + :undoc-members: diff --git a/docs/source/api/pymupdf_backend.rst b/docs/source/api/pymupdf_backend.rst new file mode 100644 index 0000000..c80ce8d --- /dev/null +++ b/docs/source/api/pymupdf_backend.rst @@ -0,0 +1,10 @@ +``document2md.pymupdf_backend`` +================================= + +Converts a born-digital PDF to Markdown by reading its own embedded text +layer with ``pymupdf4llm``, instead of running mineru's OCR. + +.. automodule:: document2md.pymupdf_backend + :members: + :private-members: + :undoc-members: diff --git a/docs/source/api/tables.rst b/docs/source/api/tables.rst new file mode 100644 index 0000000..536fa23 --- /dev/null +++ b/docs/source/api/tables.rst @@ -0,0 +1,9 @@ +``document2md.tables`` +======================== + +Turns the raw HTML tables mineru emits into Markdown tables. + +.. automodule:: document2md.tables + :members: + :private-members: + :undoc-members: diff --git a/docs/source/architecture.rst b/docs/source/architecture.rst new file mode 100644 index 0000000..c6f1bb2 --- /dev/null +++ b/docs/source/architecture.rst @@ -0,0 +1,150 @@ +Architecture +============ + +Both entry points — the ``document2md`` command line +(:py:mod:`document2md.cli`) and :py:class:`~document2md.batch.BatchConverter` +(:py:mod:`document2md.batch`) used directly from Python — go through the +same pipeline. ``BatchConverter`` resolves which backend converts a given +document (``mineru`` or ``pymupdf``; see :doc:`backends`), starting +:py:class:`~document2md.mineru_server.MineruServer` to keep a single +``mineru-api`` process warm across a batch only for the ``mineru`` backend. +:py:mod:`document2md.converter` shells out to mineru and +:py:mod:`document2md.pymupdf_backend` reads a PDF's own embedded text +layer; :py:mod:`document2md.tables` rewrites mineru's raw HTML table +fallback into Markdown tables, and :py:mod:`document2md.cutter` optionally +crops the result down to one section by title, for either backend's +output. + +.. graphviz:: + :alt: document2md's conversion pipeline, from entry points to Markdown output. + + digraph document2md_flow { + rankdir=LR; + fontname="sans-serif"; + node [fontname="sans-serif", fontsize=11, shape=box, style="rounded,filled", + fillcolor="#f4f4f4", color="#888888"]; + edge [fontname="sans-serif", fontsize=9, color="#888888"]; + + cli [label="cli.py\n(document2md command)"]; + batch [label="batch.py\nBatchConverter"]; + server [label="mineru_server.py\nMineruServer"]; + mineru [label="mineru CLI\n(external OCR/layout)", style="rounded,dashed", fillcolor="#ffffff"]; + converter [label="converter.py\nconvert_to_markdown()\nconvert_images_to_markdown()"]; + pymupdf_backend [label="pymupdf_backend.py\nhas_text_layer()\nconvert_to_markdown()"]; + tables [label="tables.py\nhtml_tables_to_markdown()"]; + cutter [label="cutter.py\ncut_markdown_by_titles()\n(optional, if titulo given)"]; + output [label="Markdown output", shape=note, style=filled, fillcolor="#ffffff"]; + + cli -> batch; + batch -> server [label="__enter__ / __exit__ (mineru backend)"]; + server -> converter [label="MINERU_API_URL", style=dashed]; + batch -> converter [label="__call__ (mineru backend)"]; + batch -> pymupdf_backend [label="__call__ (pymupdf backend)"]; + converter -> mineru [label="subprocess"]; + converter -> tables [label="rewrite HTML tables"]; + tables -> batch [label="Markdown"]; + pymupdf_backend -> batch [label="Markdown"]; + batch -> cutter [label="titulo given"]; + cutter -> output; + batch -> output [label="titulo omitted"]; + } + +The sections below walk the pipeline in the order a conversion actually +flows through it. + +``document2md.cli`` — command-line entry point +------------------------------------------------ + +The ``document2md`` console script. Parses arguments and drives one +:py:class:`~document2md.batch.BatchConverter` conversion, printing the +resolved backend and where the result was saved. See the user site's +`Command line `_ +page for its flags and worked examples; :doc:`api/cli` for the full API +reference. + +``document2md.batch`` — Python entry point +--------------------------------------------- + +:py:class:`~document2md.batch.BatchConverter` is the package's public entry +point when used from Python, re-exported off :py:mod:`document2md` itself. +As a context manager, entering it resolves the requested backend and +starts a persistent ``mineru-api`` server only when that backend is +``mineru`` (skipped if a caller further up already has one running via +``MINERU_API_URL`` — see ``document2md.mineru_server`` below); exiting +stops it. Calling it converts one document — a single PDF path, or a list +of image paths for a document spanning several scanned pages — to +Markdown. See :doc:`backends` for how resolution and dispatch work, and +the user site's `Python `_ +page for worked examples; :doc:`api/batch` for the full API reference. + +``document2md.mineru_server`` — keeping mineru-api warm +----------------------------------------------------------- + +``BatchConverter.__enter__`` starts a +:py:class:`~document2md.mineru_server.MineruServer`, which launches +``mineru-api`` as a subprocess, waits for it to report healthy, and points +every conversion in the batch at it via the ``MINERU_API_URL`` environment +variable — instead of the ``mineru`` CLI spinning up (and reloading all +layout/OCR models into) a fresh temporary server on every single +invocation. See :doc:`api/mineru_server` for the full API reference. + +``document2md.converter`` — running mineru +----------------------------------------------- + +Each ``BatchConverter.__call__``, under the ``mineru`` backend, shells out +to the ``mineru`` CLI — reusing the ``MINERU_API_URL`` server above when +set — to OCR a PDF (:py:func:`~document2md.converter.convert_to_markdown`) +or one or more scanned page images +(:py:func:`~document2md.converter.convert_images_to_markdown`), then hands +mineru's raw Markdown to ``document2md.tables`` below before writing the +result to disk. See :doc:`api/converter` for the full API reference. + +``document2md.pymupdf_backend`` — reading a PDF's own text layer +---------------------------------------------------------------------- + +Under the ``pymupdf`` backend, ``BatchConverter.__call__`` instead reads a +born-digital PDF's own embedded text layer with +`pymupdf4llm `_ +(:py:func:`~document2md.pymupdf_backend.convert_to_markdown`) — far faster +than OCR, at the cost of slightly worse structure, and with no image +extraction. See :doc:`backends` for how a PDF qualifies for this backend +and :doc:`api/pymupdf_backend` for the full API reference. + +``document2md.tables`` — HTML tables to Markdown tables +------------------------------------------------------------ + +mineru renders simple tables as Markdown but falls back to raw HTML +(``…
`` with rowspan/colspan) for anything complex; this +module rewrites those into GitHub Markdown tables so the ``mineru`` +backend's output is Markdown all the way through. See :doc:`api/tables` +for the full API reference. + +``document2md.cutter`` — cropping to a single section +------------------------------------------------------------ + +When ``BatchConverter`` is called with ``titulo``, the last step before +the Markdown is written is slicing it down to the text between this +section's title and the next section's title, as they appear in the +source document's own index — the converted text otherwise spans whatever +sections shared a page. Needing no file on disk, this is a real, +executed example rather than one marked ``# doctest: +SKIP``: + +>>> from document2md.cutter import cut_markdown_by_titles +>>> markdown = ( +... "resto de la nota anterior.\n\n" +... "## Acuerdo de regularizacion de titulos\n\n" +... "Cuerpo del acuerdo.\n\n" +... "## Norma Oficial Mexicana NOM-042-NUCL\n\n" +... "Nota siguiente, excluir.\n" +... ) +>>> cut = cut_markdown_by_titles( +... markdown, +... "Acuerdo de regularizacion de titulos", +... "Norma Oficial Mexicana NOM-042-NUCL", +... ) +>>> print(cut) +## Acuerdo de regularizacion de titulos + +Cuerpo del acuerdo. + +See :doc:`api/cutter` for the full API reference. diff --git a/docs/source/backends.rst b/docs/source/backends.rst new file mode 100644 index 0000000..5001819 --- /dev/null +++ b/docs/source/backends.rst @@ -0,0 +1,86 @@ +Backends +======== + +:py:class:`~document2md.batch.BatchConverter` and the ``document2md`` CLI +share one selector: a ``backend`` argument (``--backend`` on the CLI) +naming who converts a given document — ``"auto"`` (default), ``"mineru"`` +or ``"pymupdf"``. This page is about how that selector is implemented, and +what it takes to add a fourth backend; see the user site's +`Backends `_ +page for the `auto` policy and a comparison of the two backends that ship +today. + +How resolution and dispatch work +--------------------------------- + +Three steps, in order: + +1. **Validation, in** ``BatchConverter.__init__``. The requested name is + checked against ``document2md.batch.BACKENDS``; anything else + raises ``ValueError`` immediately, before any I/O happens. +2. **Resolution, in** ``BatchConverter.__enter__``. ``"auto"`` becomes + ``"mineru"`` when ``shutil.which("mineru")`` finds it on ``PATH``, + otherwise ``"pymupdf"``; any other requested name passes through + unchanged. The resolved name is stored as ``self.backend``. Still in + ``__enter__``, the resolved backend's own dependency is checked (each + backend's ``_require_*`` function — see below) and, only for + ``mineru``, a :py:class:`~document2md.mineru_server.MineruServer` is + started. Doing this in ``__enter__`` rather than lazily on the first + call means a missing dependency is reported before any document is + converted, not partway through a batch. +3. **Dispatch, in** ``BatchConverter.__call__``. A plain ``if``/``elif`` on + ``self.backend`` routes to the right converter function. The + ``pymupdf`` branch additionally rejects input it structurally cannot + handle — a list of page images, or a PDF that fails + :py:func:`~document2md.pymupdf_backend.has_text_layer` — by raising + ``RuntimeError`` rather than silently falling back to ``mineru``. + +The contract a backend must satisfy +------------------------------------- + +Concretely, from ``mineru``/``pymupdf`` as the two examples: + +- **A conversion function** taking an input path (or list of paths, for + ``mineru``'s scanned-page-image case) and a ``md_path``, writing + Markdown to ``md_path`` and returning nothing — + :py:func:`document2md.converter.convert_to_markdown` and + :py:func:`document2md.pymupdf_backend.convert_to_markdown`. +- **A** ``_require_*()`` **function** raising ``RuntimeError`` with an + install hint when the backend's own dependency isn't available — + :py:func:`document2md.converter._require_mineru` and + :py:func:`document2md.pymupdf_backend._require_pymupdf`. ``BatchConverter.__enter__`` + calls the right one for the resolved backend. +- **Lazy imports of the backend's own dependency**, inside functions + rather than at module level. The docs build installs the package + ``--no-deps`` (see :doc:`development`), and autodoc must still be able + to import every module even when a backend's dependency isn't + installed at all. + +Adding a backend +----------------- + +1. Write the new module (``document2md/_backend.py`` is the + existing naming pattern, though not enforced), with a conversion + function and a ``_require_*()`` function satisfying the contract + above. Import the backend's own dependency lazily, inside functions. +2. Add the new name to ``document2md.batch.BACKENDS``. +3. Extend ``BatchConverter.__enter__``'s resolution: decide what ``"auto"`` + should do when this backend is also available (mineru is preferred + over pymupdf today — see the user site's Backends page for why), call + the new ``_require_*()`` for the resolved backend, and start any + server the backend needs (most won't). +4. Extend ``BatchConverter.__call__``'s dispatch with a new branch, and + decide what it does with input it can't handle — raising + ``RuntimeError`` naming what to install or use instead, as ``pymupdf`` + does, rather than silently falling back to another backend. +5. Add the new choice to the CLI's ``--backend`` ``choices``. +6. Tests: backend validation and resolution in ``tests/test_batch.py`` + (following the existing ``mineru``/``pymupdf`` cases), CLI parsing and + the progress-line/error-message wiring in ``tests/test_cli.py``, and a + dedicated test module for the backend's own conversion function and + ``_require_*()`` check, run for real rather than mocked if the + dependency is small enough to install directly (as + ``tests/test_pymupdf_backend.py`` does). +7. Document it: a one-sentence :doc:`api/index` entry and API page, this + page's contract/dispatch description if the new backend needs + anything beyond it, and the user site's Backends and Install pages. diff --git a/docs/source/conf.py b/docs/source/conf.py index 38e05ef..ebf47f2 100644 --- a/docs/source/conf.py +++ b/docs/source/conf.py @@ -55,7 +55,6 @@ # above, plus :py:func:/:py:mod: roles on those three names in index.rst, is # all it takes. -templates_path = ["_templates"] source_suffix = ".rst" master_doc = "index" language = "en" @@ -69,15 +68,18 @@ add_module_names = False # document2md never imports mineru itself in-process — it only shells out to the -# mineru-api CLI as a subprocess (see document2md/mineru_server.py) — so autodoc -# needs no mock for it; document2md is installed with --no-deps in -# .readthedocs.yaml precisely so the heavy mineru[pipeline] install (and its -# libgl1/opencv system dependency, see .github/workflows/test.yml) never has to +# mineru-api CLI as a subprocess (see document2md/mineru_server.py) — and +# document2md/pymupdf_backend.py imports pymupdf/pymupdf4llm lazily, inside +# functions rather than at module level, so autodoc needs no mock for either; +# document2md is installed with --no-deps in .readthedocs.yaml precisely so +# neither the heavy mineru[pipeline] install (and its libgl1/opencv system +# dependency, see .github/workflows/test.yml) nor pymupdf4llm ever has to # happen for a docs build. # -- Options for HTML output ---------------------------------------------- html_theme = "furo" +html_title = "document2md" htmlhelp_basename = "document2mddoc" # -- Options for LaTeX/manual/texinfo output ------------------------------- diff --git a/docs/source/development.rst b/docs/source/development.rst new file mode 100644 index 0000000..25f939e --- /dev/null +++ b/docs/source/development.rst @@ -0,0 +1,114 @@ +Development +=========== + +This page condenses the repository's own ``CLAUDE.md``; that file is the +source of truth if the two ever disagree. + +Repository layout +------------------- + +The repository *is* the ``document2md`` package: ``pyproject.toml`` and +``document2md/`` sit at the root, alongside ``tests/`` for its pytest +suite. Two other things share the repository: + +- ``dof2md/`` — a code-less final release for the package's old PyPI name, + a *tombstone* (see below), with its own ``tests/``. +- ``website/`` — the Quarto user site published to GitHub Pages at + `ingeotec.github.io/document2md `_; + ``docs/`` is this Read the Docs developer site instead. The two have no + overlapping page: Pages is for using the package, Read the Docs for + extending it. + +Two packages are published from one repository, so the release tag +convention is ``-v`` (``document2md-v0.4.0``, +``dof2md-v0.3.0``) rather than a bare ``v*``, which wouldn't say which one. + +Running the tests +-------------------- + +.. code-block:: bash + + python -m pytest tests # document2md + python -m pytest dof2md/tests # the tombstone + +**Always two invocations, never a bare** ``pytest`` **at the repository +root.** Both suites live in a directory called ``tests``, and, more to the +point, the ``dof2md/`` container directory shadows the installed +``dof2md`` distribution as a namespace package at the repository root — so +a bare ``pytest`` would import the namespace-package ``dof2md`` silently +instead of raising, and ``dof2md/tests/test_tombstone.py`` would fail. +Pointing pytest at ``dof2md/tests`` puts ``dof2md/`` itself at the front of +``sys.path``, where the real raising module wins. ``test.yml`` runs them as +two separate jobs for the same reason. + +The suite never imports ``mineru`` — it mocks the subprocess boundary +(``document2md.converter.convert_to_markdown``/``convert_images_to_markdown``, +``document2md.batch.MineruServer``) — so a light local install +(``pip install --no-deps -e .`` plus ``requests``, ``pymupdf4llm`` and +``pytest``) runs the whole suite without the gigabytes of +``mineru[pipeline]``. ``tests/test_pymupdf_backend.py`` is the one +exception: it exercises the real ``pymupdf``/``pymupdf4llm``, building its +own PDFs on the fly with PyMuPDF, since that backend is small and core +rather than optional. + +Building the docs and the website +------------------------------------ + +.. code-block:: bash + + python -m sphinx -n -W --keep-going -b html docs/source docs/build/html + python -m sphinx -b doctest docs/source docs/build/doctest + quarto render website + +The Sphinx HTML build is strict (``-n -W``): any broken cross-reference or +other warning fails it. The doctest build runs this site's few real, +executed examples (see :doc:`architecture`'s ``cutter`` example); usage +examples that would need ``mineru`` or ``pymupdf4llm`` live on the Pages +site instead, not here, so there is little left to mark +``# doctest: +SKIP``. Quarto (pinned at 1.9.38, matching +``.github/workflows/website.yml``) needs no Python, R or Jupyter to render +``website/``, since none of its pages execute code. + +Versioning and publishing +---------------------------- + +``scripts/check_package_versions.py`` gates every pull request: a +package's ``__version__`` may be at most one release ahead of what's +published on PyPI (the next patch, or the next minor with patch reset to +0) — never equal, never further ahead. Publishing itself is a human +action: push a ``document2md-v`` (or ``dof2md-v``) tag +once ``__version__`` is bumped, with the ``TWINE`` secret set on the +repository; ``publish-pypi.yml`` refuses a tag that disagrees with +``pyproject.toml``. ``document2md`` must reach PyPI before or at the same +time as any ``nota2md`` release whose ``ocr`` extra requires it +(``document2md[mineru]>=0.4.0``), or ``pip install nota2md[ocr]`` breaks +for everyone outside the LegalIA repository. The Pages site publishes +itself, via ``website.yml``, on every push to ``main`` that touches +``website/``; Read the Docs must be imported once, by a human, at +readthedocs.org — nothing in the repository can do that. + +The ``dof2md`` tombstone +--------------------------- + +``dof2md/`` is a code-less final release for the old PyPI name: +``__version__ = "0.3.0"``, then an unconditional ``ImportError`` naming +``document2md`` and ``pip install document2md``. Three rules, each guarded +by a test: + +- ``__version__`` stays above the ``raise``, as a plain literal, so both + setuptools' ``attr:`` resolution and ``check_package_versions.py`` can + read it statically (AST, no import). +- It does not depend on ``document2md`` — a working shim would keep the + old package alive as real, maintained code. +- ``[tool.document2md] tombstone = true`` in ``dof2md/pyproject.toml`` + exempts it from the one-step-ahead rule once its final version is + published, since local and PyPI then agree forever. + +Language policy +------------------- + +Everything written into the repository is in English — identifiers, +comments, docstrings, commit messages, documentation, both sites included. +The one deliberate exception is the CLI's Spanish flags +``--titulo``/``--titulo-siguiente``, kept through the package's rename +because changing them would be a behaviour change, not a cleanup. diff --git a/docs/source/index.rst b/docs/source/index.rst index 4301f6e..c0854c3 100644 --- a/docs/source/index.rst +++ b/docs/source/index.rst @@ -7,278 +7,48 @@ .. image:: https://badge.fury.io/py/document2md.svg :target: https://badge.fury.io/py/document2md +.. image:: https://readthedocs.org/projects/document2md/badge/?version=latest + :target: https://document2md.readthedocs.io/en/latest/ + .. Registers the top-level package as a cross-reference target (it renders nothing): every module below is documented by its own `automodule`, but the package itself is only ever referred to, so without this the `:py:mod:` references to it dangle under `sphinx -n`. .. py:module:: document2md -Version |document2md_version|. This package was extracted from the `LegalIA -`_ monorepo at commit ``e1f258c``; the -packages it used to share a repository with are documented at -`legalia.readthedocs.io `_. - -:py:mod:`document2md` converts a PDF or a set of scanned page images — from -Mexico's official gazette (DOF, *Diario Oficial de la Federación*) or any -other document — into Markdown, optionally cropped down to a single note. -It wraps `mineru `_ for the OCR and -layout analysis itself; :py:mod:`document2md`'s own contribution is keeping -mineru's ``mineru-api`` server warm across a batch of documents, stitching -several scanned pages of the same note into one continuous Markdown -document, rewriting mineru's raw HTML table fallback into Markdown tables, -and cropping the result down to a single note by locating its title and the -next note's title in the OCR'd text. It has no notion of a "note"/legal -provision of its own, and no download of its own — getting a whole DOF -edition's PDF by date and edition is `dofjson.download_edicion_pdf -`_'s -job; `nota2md `_ -is what calls into :py:mod:`document2md` at all, and only -as its OCR fallback for legal provisions predating the HTML era. Both live in -the `LegalIA `_ repository and are not -dependencies of this one. - -document2md's architecture -========================== - -Both entry points below — the ``document2md`` command line (:py:mod:`document2md.cli`) -and :py:class:`~document2md.batch.BatchConverter` (:py:mod:`document2md.batch`) used directly -from Python — go through the same pipeline. :py:class:`~document2md.mineru_server.MineruServer` -keeps a single ``mineru-api`` process warm across a batch instead of paying -its startup cost per document; :py:mod:`document2md.converter` shells out to it, -:py:mod:`document2md.tables` rewrites mineru's raw HTML table fallback into -Markdown tables, and :py:mod:`document2md.cutter` optionally crops the result down -to one note by title. - -.. graphviz:: - :alt: document2md's conversion pipeline, from entry points to Markdown output. - - digraph document2md_flow { - rankdir=LR; - fontname="sans-serif"; - node [fontname="sans-serif", fontsize=11, shape=box, style="rounded,filled", - fillcolor="#f4f4f4", color="#888888"]; - edge [fontname="sans-serif", fontsize=9, color="#888888"]; - - cli [label="cli.py\n(document2md command)"]; - batch [label="batch.py\nBatchConverter"]; - server [label="mineru_server.py\nMineruServer"]; - mineru [label="mineru CLI\n(external OCR/layout)", style="rounded,dashed", fillcolor="#ffffff"]; - converter [label="converter.py\nconvert_to_markdown()\nconvert_images_to_markdown()"]; - tables [label="tables.py\nhtml_tables_to_markdown()"]; - cutter [label="cutter.py\ncut_markdown_by_titles()\n(optional, if titulo given)"]; - output [label="Markdown output", shape=note, style=filled, fillcolor="#ffffff"]; - - cli -> batch; - batch -> server [label="__enter__ / __exit__"]; - server -> converter [label="MINERU_API_URL", style=dashed]; - batch -> converter [label="__call__"]; - converter -> mineru [label="subprocess"]; - converter -> tables [label="rewrite HTML tables"]; - tables -> batch [label="Markdown"]; - batch -> cutter [label="titulo given"]; - cutter -> output; - batch -> output [label="titulo omitted"]; - } - -The sections below are ordered the way a conversion actually flows through -the package: the two entry points first, then each module in turn. Every -class and function is documented, including private/internal helpers -(leading-underscore names) — useful when extending or debugging the package, +Version |document2md_version|. :py:mod:`document2md` converts a PDF, or an +ordered set of scanned page images, into Markdown, via one of two backends +— `mineru `_ (OCR and layout +analysis) or `pymupdf4llm `_ (reading a +born-digital PDF's own embedded text layer, no OCR needed) — optionally +cropped down to a single section. This package was extracted from the +`LegalIA `_ monorepo at commit +``e1f258c``; the packages it used to share a repository with are +documented at `legalia.readthedocs.io `_. + +.. note:: + This is the **developer** documentation: architecture, backends, and + the full API reference, for extending the package itself. If you only + want to *use* document2md — install it, run it from the command line or + from Python, pick a backend — see the user site at + `ingeotec.github.io/document2md `_. + +How this documentation is organised +==================================== + +.. toctree:: + :maxdepth: 2 + + architecture + backends + development + api/index + +:doc:`architecture` walks through the conversion pipeline end to end, module +by module, in the order a document actually flows through it. +:doc:`backends` explains how backend resolution and dispatch work, and how +to add a new backend. :doc:`development` covers the repository's own +conventions — tests, docs and website builds, versioning, publishing. +:doc:`api/index` is the full API reference, one page per module, including +private/internal helpers — useful when extending or debugging the package, though they are not part of its public API and can change without notice. - -**The one documented exception to "every public symbol has a verified -example".** Entering :py:class:`~document2md.batch.BatchConverter` -starts a real ``mineru-api`` server, and calling it shells out to the real -``mineru`` CLI — neither is installed in the doctest job on purpose -(``mineru[pipeline]`` is heavy, and keeping it out is exactly why -``.readthedocs.yaml``/``test.yml`` install ``document2md`` with ``--no-deps``). -Every example below that would actually invoke -mineru is marked ``# doctest: +SKIP`` and is instead exercised for real by -``tests/test_batch.py`` and ``tests/test_cli.py``, which mock only -the mineru boundary (``document2md.converter.convert_to_markdown``/ -``convert_images_to_markdown``, ``document2md.batch.MineruServer``) and run -everything else — argument forwarding, title cropping, ``keep_pages``, -``keep_mineru_output`` — for real. ``document2md.cutter`` below, needing neither -mineru nor any file on disk, is genuinely executed. - -``document2md.cli`` — command-line entry point ----------------------------------------------- - -The ``document2md`` console script. Parses arguments and drives one -:py:class:`~document2md.batch.BatchConverter` conversion, printing where the -result was saved: - -.. code-block:: console - - $ document2md --pdf edicion.pdf --outdir output - Converting to Markdown (mineru)... - Markdown saved to: output/edicion.md - -``--filename`` defaults to the ``--pdf`` file's own name (with a ``.md`` -extension); it is required with ``--images``, since a list of scanned pages -has no single input name to derive one from: - -.. code-block:: console - - $ document2md --images nota-200-p1.jpg nota-200-p2.jpg --filename nota-200.md --outdir output - Converting to Markdown (mineru)... - Markdown saved to: output/nota-200.md - -``--titulo``/``--titulo-siguiente`` crop the result to one note; -``--keep-pages`` also keeps the uncropped conversion alongside it, as -``/.full.md``: - -.. code-block:: console - - $ document2md --pdf edicion.pdf --outdir output \ - --titulo "ACUERDO por el que se..." \ - --titulo-siguiente "DECRETO por el que se..." \ - --keep-pages - Converting to Markdown (mineru)... - Markdown saved to: output/edicion.md - -.. automodule:: document2md.cli - :members: - :private-members: - :undoc-members: - -``document2md.batch`` — Python entry point ------------------------------------------- - -:py:class:`~document2md.batch.BatchConverter` is the package's public entry point when -used from Python, re-exported off :py:mod:`document2md` itself. As a context -manager it starts a persistent ``mineru-api`` server on ``__enter__`` -(skipped if a caller further up already has one running via -``MINERU_API_URL`` — see ``document2md.mineru_server`` below) and stops it on -``__exit__``; calling it converts one document — a single PDF path, or a -list of image paths for a document spanning several scanned pages — to -Markdown: - ->>> from document2md import BatchConverter ->>> ->>> jobs = [ -... ("a.pdf", "output", "a.md"), -... (["b-p1.jpg", "b-p2.jpg"], "output", "b.md"), -... ] ->>> with BatchConverter() as convert: # doctest: +SKIP -... for path_or_paths, outdir, filename in jobs: -... convert(path_or_paths, outdir, filename) - -Passing ``titulo``/``titulo_siguiente`` crops the OCR'd Markdown down to the -text between the two titles, as they appear in the gazette's own index — -useful because a scanned edition page usually holds the tail of one note and -the head of the next. Title matching is fuzzy (OCR text rarely matches an -index title exactly), so ``min_confidence`` (default ``0.6``) sets how -confident a match has to be before it is trusted — a weaker one is treated as -not found and the crop falls back to keeping more text rather than dropping -content (see ``document2md.cutter`` below). ``keep_pages=True`` also keeps the -uncropped Markdown, as ``/.full.md``; ``keep_mineru_output=True`` -keeps mineru's own raw output instead of discarding it (see -``document2md.converter`` below): - ->>> with BatchConverter() as convert: # doctest: +SKIP -... convert( -... "edicion.pdf", "output", "nota.md", -... titulo="ACUERDO por el que se...", -... titulo_siguiente="DECRETO por el que se...", -... min_confidence=0.8, -... keep_pages=True, -... keep_mineru_output=True, -... ) - -`nota2md.legal_provisions -`_ -accepts an already-``__enter__``'d -:py:class:`~document2md.batch.BatchConverter` as its own ``converter`` parameter, so a -batch of DOF legal provisions can share the same warm server too — the way -to OCR many pre-HTML-era notes without paying mineru's startup cost once per -note (``nota2md`` lives in the `LegalIA -`_ repository and is not a dependency of -this one, so this example is skipped here as well): - ->>> import nota2md # doctest: +SKIP ->>> codigos_sin_html = [4430696, 4430697] ->>> with BatchConverter() as ins: # doctest: +SKIP -... for cod_nota in codigos_sin_html: -... nota2md.legal_provisions(cod_nota, "output", source="image", converter=ins) - -.. automodule:: document2md.batch - :members: - :private-members: - :undoc-members: - -``document2md.mineru_server`` — keeping mineru-api warm -------------------------------------------------------- - -``BatchConverter.__enter__`` starts a :py:class:`~document2md.mineru_server.MineruServer`, -which launches ``mineru-api`` as a subprocess, waits for it to report -healthy, and points every conversion in the batch at it via the -``MINERU_API_URL`` environment variable — instead of the ``mineru`` CLI -spinning up (and reloading all layout/OCR models into) a fresh temporary -server on every single invocation. - -.. automodule:: document2md.mineru_server - :members: - :private-members: - :undoc-members: - -``document2md.converter`` — running mineru ------------------------------------------- - -Each ``BatchConverter.__call__`` shells out to the ``mineru`` CLI — -reusing the ``MINERU_API_URL`` server above when set — to OCR a PDF -(:py:func:`~document2md.converter.convert_to_markdown`) or one or more scanned -page images (:py:func:`~document2md.converter.convert_images_to_markdown`), then -hands mineru's raw Markdown to ``document2md.tables`` below before writing the -result to disk. - -.. automodule:: document2md.converter - :members: - :private-members: - :undoc-members: - -``document2md.tables`` — HTML tables to Markdown tables -------------------------------------------------------- - -mineru renders simple tables as Markdown but falls back to raw HTML -(``…
`` with rowspan/colspan) for anything complex; this -module rewrites those into GitHub Markdown tables so both conversion -functions above return Markdown all the way through. - -.. automodule:: document2md.tables - :members: - :private-members: - :undoc-members: - -``document2md.cutter`` — cropping to a single note --------------------------------------------------- - -When ``BatchConverter`` is called with ``titulo``, the last step before the -Markdown is written is slicing it down to the text between this note's title -and the next note's title, as they appear in the gazette's own per-day -index — the OCR'd text otherwise spans whatever notes shared that scanned -page. Needing neither mineru nor a file on disk, this is the one example on -this page that runs for real rather than under ``# doctest: +SKIP``: - ->>> from document2md.cutter import cut_markdown_by_titles ->>> markdown = ( -... "resto de la nota anterior.\n\n" -... "## Acuerdo de regularizacion de titulos\n\n" -... "Cuerpo del acuerdo.\n\n" -... "## Norma Oficial Mexicana NOM-042-NUCL\n\n" -... "Nota siguiente, excluir.\n" -... ) ->>> cut = cut_markdown_by_titles( -... markdown, -... "Acuerdo de regularizacion de titulos", -... "Norma Oficial Mexicana NOM-042-NUCL", -... ) ->>> print(cut) -## Acuerdo de regularizacion de titulos - -Cuerpo del acuerdo. - -.. automodule:: document2md.cutter - :members: - :private-members: - :undoc-members: diff --git a/document2md/__init__.py b/document2md/__init__.py index 24bacd5..7528b2d 100644 --- a/document2md/__init__.py +++ b/document2md/__init__.py @@ -1,5 +1,5 @@ from document2md.batch import BatchConverter -__version__ = "0.3.0" +__version__ = "0.4.0" __all__ = ["BatchConverter"] diff --git a/document2md/batch.py b/document2md/batch.py index 0118218..27d671a 100644 --- a/document2md/batch.py +++ b/document2md/batch.py @@ -11,23 +11,48 @@ list of paths), an output directory, and an output filename. Whatever calls this decides what those mean (a DOF legal provision, or anything else). """ +import shutil from pathlib import Path from document2md import converter as _converter +from document2md import pymupdf_backend as _pymupdf_backend from document2md.converter import DEFAULT_TIMEOUT_SECONDS from document2md.cutter import cut_markdown_by_titles from document2md.mineru_server import ENV_VAR as _MINERU_API_URL_ENV_VAR from document2md.mineru_server import MineruServer +# Backend names accepted by BatchConverter(backend=...) and --backend. +# "auto" resolves to a concrete backend in __enter__: "mineru" when the +# mineru CLI is on PATH, "pymupdf" otherwise (see __enter__). +BACKENDS = ("auto", "mineru", "pymupdf") + +_MINERU_INSTALL_HINT = 'Install the mineru backend with: pip install "document2md[mineru]"' + class BatchConverter: - """Context manager: `__enter__` starts a persistent `mineru-api` server + """Context manager: `__enter__` resolves the requested `backend`, starts + a persistent `mineru-api` server if the resolved backend needs one (skipped if a caller further up already has one running via - MINERU_API_URL) and returns `self`, callable once per document; + MINERU_API_URL), and returns `self`, callable once per document; `__exit__` stops it. Calling it converts one document — a single PDF path, or a list of image paths for a document spanning several scanned pages — to Markdown, written to `outdir/filename`. + `backend` names who is responsible for the conversion: `"auto"` + (default), `"mineru"` or `"pymupdf"`; any other value raises + `ValueError`. `"auto"` resolves to `"mineru"` when the `mineru` CLI is on + `PATH`, otherwise to `"pymupdf"` — mineru remains the reference backend + where it is installed; the light backend only makes installing it + optional. The resolved name is stored as `self.backend` once `__enter__` + has run. + + `"pymupdf"` reads a PDF's own embedded text layer instead of running + OCR — far faster, at the cost of slightly worse structure — and raises + `RuntimeError` (naming the `mineru` extra) on input it cannot handle: a + list of page images (which always need OCR), or a PDF without enough of + a text layer (see `document2md.pymupdf_backend.has_text_layer`). There is + no silent fallback to mineru. + `titulo`/`titulo_siguiente`, when given, slice the OCR'd Markdown down to the text between their two boundaries (see document2md.cutter.cut_markdown_by_titles) — e.g. a DOF legal provision's own @@ -37,19 +62,42 @@ class BatchConverter: what was asked for. """ - def __init__(self): - """Create an unstarted converter; call `__enter__` (or use as a - context manager) before calling it.""" + def __init__(self, backend: str = "auto"): + """Create an unstarted converter for the requested `backend` + (`"auto"`, `"mineru"` or `"pymupdf"`; anything else raises + `ValueError`); call `__enter__` (or use as a context manager) before + calling it.""" + if backend not in BACKENDS: + raise ValueError( + f"Unknown backend {backend!r}; accepted values: {', '.join(BACKENDS)}" + ) + self._requested_backend = backend + self.backend: str | None = None self._server: MineruServer | None = None def __enter__(self) -> "BatchConverter": - """Start a persistent `mineru-api` server, unless one is already - reachable via MINERU_API_URL, and return `self`.""" + """Resolve the requested backend (storing it as `self.backend`), + start a persistent `mineru-api` server if it needs one — unless one + is already reachable via MINERU_API_URL — and return `self`. + + `"auto"` resolves to `"mineru"` when the `mineru` CLI is on `PATH`, + otherwise to `"pymupdf"`. Raises RuntimeError if the resolved + backend's dependency isn't installed; this happens here, before any + document is converted, rather than on the first call.""" import os - if _MINERU_API_URL_ENV_VAR not in os.environ: - self._server = MineruServer() - self._server.start() + if self._requested_backend == "auto": + self.backend = "mineru" if shutil.which("mineru") is not None else "pymupdf" + else: + self.backend = self._requested_backend + + if self.backend == "mineru": + _converter._require_mineru() + if _MINERU_API_URL_ENV_VAR not in os.environ: + self._server = MineruServer() + self._server.start() + else: + _pymupdf_backend._require_pymupdf() return self def __exit__(self, exc_type, exc, tb) -> None: @@ -79,12 +127,32 @@ def __call__( forwarded to `cutter.cut_markdown_by_titles` to crop the result down to a single note; left as `None` (the default), the whole conversion is kept as-is. `keep_mineru_output` and `timeout` are forwarded to - `converter.convert_to_markdown`/`convert_images_to_markdown`.""" + `converter.convert_to_markdown`/`convert_images_to_markdown` for the + `mineru` backend; both are accepted but ignored for `pymupdf`, which + needs neither. + + Under the `pymupdf` backend, a list of page images (which always + need OCR) or a PDF without enough of an embedded text layer (see + `pymupdf_backend.has_text_layer`) raise `RuntimeError` naming the + `mineru` extra, rather than silently falling back to it.""" outdir = Path(outdir) outdir.mkdir(parents=True, exist_ok=True) dest = outdir / filename + is_image_list = isinstance(path_or_paths, (list, tuple)) - if isinstance(path_or_paths, (list, tuple)): + if self.backend == "pymupdf": + if is_image_list: + raise RuntimeError( + f"Scanned page images always need OCR. {_MINERU_INSTALL_HINT}" + ) + pdf_path = Path(path_or_paths) + if not _pymupdf_backend.has_text_layer(pdf_path): + raise RuntimeError( + f"{pdf_path} has no embedded text layer and needs OCR. " + f"{_MINERU_INSTALL_HINT}" + ) + _pymupdf_backend.convert_to_markdown(pdf_path, dest) + elif is_image_list: _converter.convert_images_to_markdown( [Path(p) for p in path_or_paths], dest, timeout=timeout, keep_mineru_output=keep_mineru_output, diff --git a/document2md/cli.py b/document2md/cli.py index 8fe0de8..5f35a79 100644 --- a/document2md/cli.py +++ b/document2md/cli.py @@ -55,6 +55,12 @@ def parse_args(argv=None): help="Keep mineru's raw output (layout/model JSON, rendered PDFs...) in " "/_mineru/ instead of discarding it", ) + parser.add_argument( + "--backend", choices=("auto", "mineru", "pymupdf"), default="auto", + help="Conversion backend to use (default: auto, which resolves to mineru " + "when it's on PATH, otherwise pymupdf, reading a PDF's own embedded text " + "layer without OCR)", + ) return parser.parse_args(argv) @@ -62,7 +68,8 @@ def main(argv=None): """Entry point for the `document2md` console script: parse arguments, run one `BatchConverter` conversion, and print where the Markdown was saved. Exits with an error message (no traceback) on a missing/ambiguous input - source or a mineru timeout.""" + source, a mineru timeout, or a backend whose dependency isn't + installed.""" args = parse_args(argv) sources_given = sum(x is not None for x in (args.pdf, args.images)) @@ -88,9 +95,9 @@ def main(argv=None): md_filename = args.filename path_or_paths = image_paths - print("Converting to Markdown (mineru)...") try: - with BatchConverter() as convert: + with BatchConverter(backend=args.backend) as convert: + print(f"Converting to Markdown ({convert.backend})...") md_path = convert( path_or_paths, outdir, md_filename, args.titulo, args.titulo_siguiente, min_confidence=args.min_confidence, @@ -102,6 +109,8 @@ def main(argv=None): f"Conversion timed out after {DEFAULT_TIMEOUT_SECONDS}s. " "This document may be unusually large." ) + except RuntimeError as err: + sys.exit(str(err)) print(f"Markdown saved to: {md_path}") diff --git a/document2md/converter.py b/document2md/converter.py index 0657757..8e44a5a 100644 --- a/document2md/converter.py +++ b/document2md/converter.py @@ -45,8 +45,8 @@ def _require_mineru() -> None: isn't on PATH, instead of letting `subprocess.run` fail opaquely.""" if shutil.which("mineru") is None: raise RuntimeError( - "'mineru' is required to convert documents but isn't installed. " - "Install document2md's dependencies: pip install document2md" + "'mineru' is required for the mineru backend but isn't installed. " + 'Install it with: pip install "document2md[mineru]"' ) diff --git a/document2md/pymupdf_backend.py b/document2md/pymupdf_backend.py new file mode 100644 index 0000000..37b570c --- /dev/null +++ b/document2md/pymupdf_backend.py @@ -0,0 +1,70 @@ +"""Convert a born-digital PDF (one that already carries an embedded text +layer) to Markdown by reading that text layer directly with `pymupdf4llm`, +instead of running mineru's OCR/layout models on it. Orders of magnitude +faster than the mineru backend and needs no models, at the cost of slightly +worse structure; mineru remains the reference backend where it is installed +(see document2md.batch.BatchConverter's `auto` resolution). + +PyMuPDF and pymupdf4llm are AGPL-3.0 licensed; document2md itself stays +Apache-2.0 (see the README's Install section). + +`pymupdf`/`pymupdf4llm` are imported inside the functions below, never at +module level, so this module can still be imported (for autodoc, or by +BatchConverter probing `has_text_layer`) when the package is installed with +`--no-deps` and neither is present. +""" +from pathlib import Path + +# A page counts as text-bearing when its extracted text has at least this +# many non-whitespace characters. +MIN_CHARS_PER_PAGE = 50 + +# The PDF as a whole counts as having a text layer when at least this +# fraction of its pages are text-bearing — tolerating a few scanned inserts +# in an otherwise born-digital document. +MIN_TEXT_PAGE_FRACTION = 0.8 + + +def _require_pymupdf() -> None: + """Raise RuntimeError with an actionable message if `pymupdf4llm` isn't + installed, instead of letting the `import` inside convert_to_markdown + fail opaquely.""" + import importlib.util + + if importlib.util.find_spec("pymupdf4llm") is None: + raise RuntimeError( + "'pymupdf4llm' is required for the pymupdf backend but isn't " + "installed. Install it with: pip install document2md" + ) + + +def has_text_layer(pdf_path: Path) -> bool: + """Whether `pdf_path` is born-digital enough for the pymupdf backend: + at least MIN_TEXT_PAGE_FRACTION of its pages each have at least + MIN_CHARS_PER_PAGE non-whitespace characters of extractable text. + A fully scanned PDF fails; a mostly-digital document with a few scanned + inserts still passes.""" + import pymupdf + + with pymupdf.open(pdf_path) as doc: + if doc.page_count == 0: + return False + text_pages = sum( + 1 for page in doc + if len("".join(page.get_text().split())) >= MIN_CHARS_PER_PAGE + ) + return text_pages / doc.page_count >= MIN_TEXT_PAGE_FRACTION + + +def convert_to_markdown(pdf_path: Path, md_path: Path) -> None: + """Convert a born-digital PDF to Markdown by extracting its embedded + text layer with pymupdf4llm — headings inferred from font sizes, + Markdown tables via PyMuPDF's table finder, reading order handled — + instead of running mineru's OCR/layout models on it. + + Images are not extracted (no `_images/` directory is produced), + unlike the mineru backend.""" + import pymupdf4llm + + md_text = pymupdf4llm.to_markdown(str(pdf_path), write_images=False) + md_path.write_text(md_text + "\n", encoding="utf-8") diff --git a/pyproject.toml b/pyproject.toml index 27fa63a..bf6edbd 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -17,11 +17,12 @@ classifiers = [ ] dependencies = [ "requests>=2.31", - "mineru[pipeline]", + "pymupdf4llm>=1.28", ] dynamic = ['version'] [project.optional-dependencies] +mineru = ["mineru[pipeline]"] test = ["pytest>=7.0"] [project.urls] diff --git a/tests/test_batch.py b/tests/test_batch.py index 121db28..91e8f27 100644 --- a/tests/test_batch.py +++ b/tests/test_batch.py @@ -11,8 +11,18 @@ class TestBatchConverterServerLifecycle(unittest.TestCase): def setUp(self): self._original_env = os.environ.pop(MINERU_API_URL_ENV_VAR, None) + # __enter__ now checks the mineru backend's dependency, and (for + # "auto") whether mineru is on PATH, before starting the server; + # these tests are about the server lifecycle, not that check, so + # pretend mineru is always present and satisfied. + self._require_mineru_patcher = patch("document2md.batch._converter._require_mineru") + self._require_mineru_patcher.start() + self._which_patcher = patch("document2md.batch.shutil.which", return_value="/usr/bin/mineru") + self._which_patcher.start() def tearDown(self): + self._require_mineru_patcher.stop() + self._which_patcher.stop() if self._original_env is None: os.environ.pop(MINERU_API_URL_ENV_VAR, None) else: @@ -36,6 +46,75 @@ def test_skips_starting_a_server_when_caller_already_has_one(self, mock_server_c mock_server_cls.assert_not_called() +class TestBatchConverterBackend(unittest.TestCase): + def setUp(self): + self._original_env = os.environ.pop(MINERU_API_URL_ENV_VAR, None) + + def tearDown(self): + if self._original_env is None: + os.environ.pop(MINERU_API_URL_ENV_VAR, None) + else: + os.environ[MINERU_API_URL_ENV_VAR] = self._original_env + + def test_unknown_backend_raises_value_error(self): + with self.assertRaises(ValueError): + BatchConverter(backend="nope") + + @patch("document2md.batch.MineruServer") + @patch("document2md.batch._converter._require_mineru") + @patch("document2md.batch.shutil.which", return_value="/usr/bin/mineru") + def test_auto_resolves_to_mineru_when_on_path(self, mock_which, mock_require_mineru, mock_server_cls): + with BatchConverter(backend="auto") as convert: + self.assertEqual(convert.backend, "mineru") + mock_server_cls.return_value.start.assert_called_once() + + @patch("document2md.batch._pymupdf_backend._require_pymupdf") + @patch("document2md.batch.shutil.which", return_value=None) + def test_auto_resolves_to_pymupdf_when_mineru_not_on_path(self, mock_which, mock_require_pymupdf): + with BatchConverter(backend="auto") as convert: + self.assertEqual(convert.backend, "pymupdf") + + @patch("document2md.batch.MineruServer") + @patch("document2md.batch._converter._require_mineru") + def test_explicit_mineru_backend(self, mock_require_mineru, mock_server_cls): + with BatchConverter(backend="mineru") as convert: + self.assertEqual(convert.backend, "mineru") + + @patch( + "document2md.batch._converter._require_mineru", + side_effect=RuntimeError( + "'mineru' is required for the mineru backend but isn't installed. " + 'Install it with: pip install "document2md[mineru]"' + ), + ) + def test_enter_raises_when_mineru_is_missing(self, mock_require_mineru): + with self.assertRaises(RuntimeError) as ctx: + with BatchConverter(backend="mineru"): + pass + + self.assertIn("document2md[mineru]", str(ctx.exception)) + + @patch("document2md.batch.MineruServer") + def test_explicit_pymupdf_backend_never_starts_a_server(self, mock_server_cls): + with BatchConverter(backend="pymupdf") as convert: + self.assertEqual(convert.backend, "pymupdf") + mock_server_cls.assert_not_called() + + @patch( + "document2md.batch._pymupdf_backend._require_pymupdf", + side_effect=RuntimeError( + "'pymupdf4llm' is required for the pymupdf backend but isn't installed. " + "Install it with: pip install document2md" + ), + ) + def test_enter_raises_when_pymupdf_is_missing(self, mock_require_pymupdf): + with self.assertRaises(RuntimeError) as ctx: + with BatchConverter(backend="pymupdf"): + pass + + self.assertIn("pymupdf4llm", str(ctx.exception)) + + class TestBatchConverterCall(unittest.TestCase): def setUp(self): self.tmpdir = tempfile.TemporaryDirectory() @@ -151,5 +230,69 @@ def test_creates_outdir_if_missing(self, mock_convert): self.assertTrue(dest.exists()) +class TestBatchConverterCallPymupdf(unittest.TestCase): + def setUp(self): + self.tmpdir = tempfile.TemporaryDirectory() + self.outdir = Path(self.tmpdir.name) + self.pdf_path = self.outdir / "nota-300.pdf" + self.pdf_path.write_bytes(b"%PDF-1.4 fake") + self.image_paths = [self.outdir / "nota-200-p1.jpg", self.outdir / "nota-200-p2.jpg"] + for p in self.image_paths: + p.write_bytes(b"\xff\xd8\xff fake jpeg") + # __call__ dispatches on self.backend, normally set by __enter__; + # set it directly to exercise __call__ without going through the + # pymupdf dependency check in __enter__. + self.converter = BatchConverter() + self.converter.backend = "pymupdf" + + def tearDown(self): + self.tmpdir.cleanup() + + def test_image_list_raises_ocr_hint(self): + with self.assertRaises(RuntimeError) as ctx: + self.converter(self.image_paths, self.outdir, "nota-200.md") + self.assertIn("document2md[mineru]", str(ctx.exception)) + + @patch("document2md.batch._pymupdf_backend.has_text_layer", return_value=False) + def test_pdf_without_text_layer_raises_ocr_hint(self, mock_has_text_layer): + with self.assertRaises(RuntimeError) as ctx: + self.converter(self.pdf_path, self.outdir, "nota-300.md") + + self.assertIn("no embedded text layer", str(ctx.exception)) + self.assertIn("document2md[mineru]", str(ctx.exception)) + + @patch("document2md.batch._pymupdf_backend.convert_to_markdown") + @patch("document2md.batch._pymupdf_backend.has_text_layer", return_value=True) + def test_pdf_with_text_layer_is_converted(self, mock_has_text_layer, mock_convert): + mock_convert.side_effect = lambda pdf, md: md.write_text("texto", encoding="utf-8") + + dest = self.converter(self.pdf_path, self.outdir, "nota-300.md") + + self.assertEqual(dest, self.outdir / "nota-300.md") + mock_convert.assert_called_once_with(self.pdf_path, dest) + + @patch("document2md.batch._pymupdf_backend.convert_to_markdown") + @patch("document2md.batch._pymupdf_backend.has_text_layer", return_value=True) + def test_titulo_cropping_still_applies(self, mock_has_text_layer, mock_convert): + mock_convert.side_effect = lambda pdf, md: md.write_text( + "resto de la nota anterior.\n\n" + "## Acuerdo de regularización de títulos\n\n" + "Cuerpo del acuerdo.\n\n" + "## Norma Oficial Mexicana NOM-042-NUCL\n\n" + "Nota siguiente, excluir.\n", + encoding="utf-8", + ) + + dest = self.converter( + self.pdf_path, self.outdir, "nota-300.md", + "Acuerdo de regularización de títulos", + "Norma Oficial Mexicana NOM-042-NUCL", + ) + + text = dest.read_text(encoding="utf-8") + self.assertTrue(text.startswith("## Acuerdo de regularización")) + self.assertNotIn("NOM-042-NUCL", text) + + if __name__ == "__main__": unittest.main() diff --git a/tests/test_cli.py b/tests/test_cli.py index 8a8c029..b66e980 100644 --- a/tests/test_cli.py +++ b/tests/test_cli.py @@ -1,3 +1,5 @@ +import contextlib +import io import subprocess import tempfile import unittest @@ -19,6 +21,7 @@ def test_defaults(self): self.assertEqual(args.min_confidence, 0.6) self.assertFalse(args.keep_pages) self.assertFalse(args.keep_mineru_output) + self.assertEqual(args.backend, "auto") def test_images_and_filename_flags(self): args = parse_args(["--images", "a.jpg", "b.jpg", "--filename", "out.md"]) @@ -42,6 +45,18 @@ def test_title_cropping_flags(self): self.assertEqual(args.min_confidence, 0.8) self.assertTrue(args.keep_pages) + def test_backend_flag(self): + args = parse_args(["--pdf", "edicion.pdf", "--backend", "mineru"]) + self.assertEqual(args.backend, "mineru") + + def test_backend_flag_pymupdf(self): + args = parse_args(["--pdf", "edicion.pdf", "--backend", "pymupdf"]) + self.assertEqual(args.backend, "pymupdf") + + def test_unknown_backend_rejected(self): + with self.assertRaises(SystemExit): + parse_args(["--pdf", "edicion.pdf", "--backend", "nope"]) + class TestMain(unittest.TestCase): def test_main_requires_exactly_one_input_source(self): @@ -141,6 +156,60 @@ def test_main_exits_clearly_when_conversion_times_out(self, mock_batch_converter with self.assertRaises(SystemExit): main(["--pdf", str(pdf_path), "--outdir", tmpdir]) + @patch("document2md.cli.BatchConverter") + def test_main_passes_backend_flag(self, mock_batch_converter): + with tempfile.TemporaryDirectory() as tmpdir: + pdf_path = Path(tmpdir) / "edicion.pdf" + pdf_path.write_bytes(b"%PDF-1.4") + + main(["--pdf", str(pdf_path), "--outdir", tmpdir, "--backend", "mineru"]) + + mock_batch_converter.assert_called_once_with(backend="mineru") + + @patch("document2md.cli.BatchConverter") + def test_main_exits_clearly_when_backend_dependency_missing(self, mock_batch_converter): + mock_batch_converter.return_value.__enter__.side_effect = RuntimeError( + "'mineru' is required for the mineru backend but isn't installed. " + 'Install it with: pip install "document2md[mineru]"' + ) + with tempfile.TemporaryDirectory() as tmpdir: + pdf_path = Path(tmpdir) / "edicion.pdf" + pdf_path.write_bytes(b"%PDF-1.4") + with self.assertRaises(SystemExit) as ctx: + main(["--pdf", str(pdf_path), "--outdir", tmpdir]) + + self.assertIn("document2md[mineru]", str(ctx.exception)) + + @patch("document2md.cli.BatchConverter") + def test_main_prints_resolved_pymupdf_backend(self, mock_batch_converter): + mock_convert = mock_batch_converter.return_value.__enter__.return_value + mock_convert.backend = "pymupdf" + with tempfile.TemporaryDirectory() as tmpdir: + pdf_path = Path(tmpdir) / "edicion.pdf" + pdf_path.write_bytes(b"%PDF-1.4") + + out = io.StringIO() + with contextlib.redirect_stdout(out): + main(["--pdf", str(pdf_path), "--outdir", tmpdir, "--backend", "auto"]) + + self.assertIn("Converting to Markdown (pymupdf)...", out.getvalue()) + + @patch("document2md.cli.BatchConverter") + def test_main_exits_clearly_when_pdf_has_no_text_layer(self, mock_batch_converter): + mock_convert = mock_batch_converter.return_value.__enter__.return_value + mock_convert.side_effect = RuntimeError( + "edicion.pdf has no embedded text layer and needs OCR. " + 'Install the mineru backend with: pip install "document2md[mineru]"' + ) + with tempfile.TemporaryDirectory() as tmpdir: + pdf_path = Path(tmpdir) / "edicion.pdf" + pdf_path.write_bytes(b"%PDF-1.4") + with self.assertRaises(SystemExit) as ctx: + main(["--pdf", str(pdf_path), "--outdir", tmpdir, "--backend", "pymupdf"]) + + self.assertIn("no embedded text layer", str(ctx.exception)) + self.assertIn("document2md[mineru]", str(ctx.exception)) + if __name__ == "__main__": unittest.main() diff --git a/tests/test_converter.py b/tests/test_converter.py index f78eaf9..5779c8c 100644 --- a/tests/test_converter.py +++ b/tests/test_converter.py @@ -87,8 +87,9 @@ def test_relocates_images_next_to_output(self, mock_which, mock_run): @patch("document2md.converter.shutil.which", return_value=None) def test_raises_clear_error_when_mineru_missing(self, mock_which): - with self.assertRaises(RuntimeError): + with self.assertRaises(RuntimeError) as ctx: convert_to_markdown(self.pdf_path, self.md_path) + self.assertIn("document2md[mineru]", str(ctx.exception)) @patch("document2md.converter.subprocess.run", side_effect=_fake_mineru_run_with_html_table) @patch("document2md.converter.shutil.which", return_value="/usr/local/bin/mineru") @@ -219,8 +220,9 @@ def test_rejects_empty_image_list(self, mock_which): @patch("document2md.converter.shutil.which", return_value=None) def test_raises_when_mineru_missing(self, mock_which): - with self.assertRaises(RuntimeError): + with self.assertRaises(RuntimeError) as ctx: convert_images_to_markdown(self.images, self.md_path) + self.assertIn("document2md[mineru]", str(ctx.exception)) @patch("document2md.converter.subprocess.run", side_effect=_fake_mineru_run_text_only) @patch("document2md.converter.shutil.which", return_value="/usr/local/bin/mineru") diff --git a/tests/test_docs.py b/tests/test_docs.py new file mode 100644 index 0000000..3aad5dd --- /dev/null +++ b/tests/test_docs.py @@ -0,0 +1,35 @@ +import re +import unittest +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[1] + +_AUTOMODULE_RE = re.compile(r"^\.\.\s+automodule::\s+(\S+)\s*$", re.MULTILINE) + + +class TestEveryModuleHasExactlyOneAutomodulePage(unittest.TestCase): + def test_every_module_is_documented_exactly_once(self): + module_names = sorted( + f"document2md.{path.stem}" + for path in (ROOT / "document2md").glob("*.py") + if path.stem != "__init__" + ) + self.assertTrue(module_names, "expected at least one document2md/*.py module") + + rst_text = "\n".join( + path.read_text(encoding="utf-8") + for path in (ROOT / "docs" / "source").rglob("*.rst") + ) + documented = _AUTOMODULE_RE.findall(rst_text) + + for name in module_names: + occurrences = documented.count(name) + self.assertEqual( + occurrences, 1, + f"{name} is named by {occurrences} automodule directives under " + "docs/source/, expected exactly 1", + ) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_pymupdf_backend.py b/tests/test_pymupdf_backend.py new file mode 100644 index 0000000..cad6af7 --- /dev/null +++ b/tests/test_pymupdf_backend.py @@ -0,0 +1,87 @@ +import tempfile +import unittest +from pathlib import Path +from unittest.mock import patch + +import pymupdf + +from document2md.pymupdf_backend import ( + _require_pymupdf, + convert_to_markdown, + has_text_layer, +) + +# A paragraph long enough to clear MIN_CHARS_PER_PAGE (50 non-whitespace chars). +_PARAGRAPH = "This page carries real embedded text, not a scanned image of a page." + + +def _make_pdf(path: Path, texts: list[str | None]) -> None: + """Build a PDF at `path` with one page per entry in `texts`: a string + inserts that text on the page, `None` leaves it blank.""" + doc = pymupdf.open() + for text in texts: + page = doc.new_page() + if text is not None: + page.insert_text((72, 72), text) + doc.save(path) + doc.close() + + +class TestHasTextLayer(unittest.TestCase): + def setUp(self): + self.tmpdir = tempfile.TemporaryDirectory() + self.pdf_path = Path(self.tmpdir.name) / "doc.pdf" + + def tearDown(self): + self.tmpdir.cleanup() + + def test_true_when_every_page_has_text(self): + _make_pdf(self.pdf_path, [_PARAGRAPH, _PARAGRAPH, _PARAGRAPH]) + self.assertTrue(has_text_layer(self.pdf_path)) + + def test_false_when_every_page_is_blank(self): + _make_pdf(self.pdf_path, [None, None, None]) + self.assertFalse(has_text_layer(self.pdf_path)) + + def test_true_at_the_08_threshold(self): + # 4/5 = 0.8, exactly the MIN_TEXT_PAGE_FRACTION threshold. + _make_pdf(self.pdf_path, [_PARAGRAPH, _PARAGRAPH, _PARAGRAPH, _PARAGRAPH, None]) + self.assertTrue(has_text_layer(self.pdf_path)) + + def test_false_below_the_08_threshold(self): + # 3/5 = 0.6, below the threshold. + _make_pdf(self.pdf_path, [_PARAGRAPH, _PARAGRAPH, _PARAGRAPH, None, None]) + self.assertFalse(has_text_layer(self.pdf_path)) + + +class TestConvertToMarkdown(unittest.TestCase): + def setUp(self): + self.tmpdir = tempfile.TemporaryDirectory() + self.pdf_path = Path(self.tmpdir.name) / "doc.pdf" + self.md_path = Path(self.tmpdir.name) / "doc.md" + + def tearDown(self): + self.tmpdir.cleanup() + + def test_writes_non_empty_markdown_with_the_inserted_text(self): + _make_pdf(self.pdf_path, [_PARAGRAPH]) + convert_to_markdown(self.pdf_path, self.md_path) + + text = self.md_path.read_text(encoding="utf-8") + self.assertIn("embedded text", text) + self.assertTrue(text.endswith("\n")) + + +class TestRequirePymupdf(unittest.TestCase): + def test_raises_when_pymupdf4llm_missing(self): + with patch("importlib.util.find_spec", return_value=None): + with self.assertRaises(RuntimeError) as ctx: + _require_pymupdf() + self.assertIn("pip install document2md", str(ctx.exception)) + + def test_does_not_raise_when_installed(self): + _require_pymupdf() + + +if __name__ == "__main__": + unittest.main() diff --git a/website/.gitignore b/website/.gitignore new file mode 100644 index 0000000..135bda2 --- /dev/null +++ b/website/.gitignore @@ -0,0 +1,6 @@ +/.quarto/ +/_site/ +**/*_files/ +output/ + +**/*.quarto_ipynb diff --git a/website/_quarto.yml b/website/_quarto.yml new file mode 100644 index 0000000..3e59703 --- /dev/null +++ b/website/_quarto.yml @@ -0,0 +1,43 @@ +project: + type: website + output-dir: _site + +lang: en + +execute: + enabled: false + +website: + title: "document2md" + site-url: https://ingeotec.github.io/document2md/ + repo-url: https://github.com/INGEOTEC/document2md + navbar: + background: "#f9f8f4" + left: + - href: index.qmd + text: Home + - href: pages/install.qmd + text: Install + - href: pages/cli.qmd + text: Command line + - href: pages/python.qmd + text: Python + - href: pages/backends.qmd + text: Backends + - href: https://document2md.readthedocs.io/ + text: Developer docs + right: + - icon: github + href: https://github.com/INGEOTEC/document2md + aria-label: GitHub + page-footer: + center: > + document2md · [INGEOTEC](https://github.com/INGEOTEC) · + Code under the Apache 2.0 license + +format: + html: + theme: [litera, style.scss] + toc: true + number-sections: false + link-external-newwindow: true diff --git a/website/index.qmd b/website/index.qmd new file mode 100644 index 0000000..4a56bce --- /dev/null +++ b/website/index.qmd @@ -0,0 +1,72 @@ +--- +title: "document2md" +subtitle: "PDFs and scanned pages to Markdown" +toc: false +number-sections: false +--- + +::: {.home-lead} +**document2md** converts a PDF, or an ordered set of scanned page images, +into Markdown — any document, such as an edition of Mexico's official +gazette (DOF, *Diario Oficial de la Federación*) — optionally cropped down +to a single note. +::: + +It is a wrapper around [mineru](https://github.com/opendatalab/MinerU) for +OCR and layout analysis on scanned pages, plus a light backend built on +[pymupdf4llm](https://github.com/pymupdf/RAG) that reads a born-digital +PDF's own embedded text layer instead, needing no OCR at all. On top of +either backend, document2md keeps mineru's own server warm across a batch +of documents, stitches several scanned pages of the same note into one +continuous Markdown document, rewrites the raw HTML tables mineru falls +back to into Markdown tables, and crops the result down to a single note by +title. It has no notion of a "note" or a legal provision of its own, and it +downloads nothing — getting the PDF or the page images onto disk is the +caller's problem. + +::: {.fronts} +::: {.front} +### Any PDF or scan + +Born-digital PDFs convert through their own text layer; scanned PDFs and +page images convert through mineru's OCR. +::: +::: {.front} +### One warm server + +Batch conversion keeps mineru's own server loaded once across many +documents, instead of paying its startup cost per document. +::: +::: {.front} +### Cropped to one section + +`--titulo`/`--titulo-siguiente` cut the result down to a single note by +locating its title and the next note's title in the converted text. +::: +::: + +## Quickstart + +```bash +pip install document2md # born-digital PDFs, no OCR +pip install "document2md[mineru]" # + OCR for scanned pages +``` + +```bash +document2md --pdf edicion.pdf --outdir output +``` + +See [Install](pages/install.qmd) for what each command gets you, +[Command line](pages/cli.qmd) and [Python](pages/python.qmd) for the two +ways to run a conversion, and [Backends](pages/backends.qmd) for how +`auto` picks between mineru and pymupdf. Extending the package itself is +documented separately, on the +[developer docs](https://document2md.readthedocs.io/). + +## Colophon {.unnumbered} + +This site was built by the document2md maintainers together with Claude, +Anthropic's coding assistant, through +[Claude Code](https://claude.com/claude-code): the assistant helped set up +the Quarto site and draft this page, with the authors reviewing and +validating every change before it was committed. diff --git a/website/pages/backends.qmd b/website/pages/backends.qmd new file mode 100644 index 0000000..275cade --- /dev/null +++ b/website/pages/backends.qmd @@ -0,0 +1,66 @@ +--- +title: "Backends" +--- + +document2md converts through one of two backends: `mineru` (OCR and +layout analysis) or `pymupdf` (reading a PDF's own embedded text layer, +no OCR). `--backend`/`BatchConverter(backend=...)` picks between them, +defaulting to `"auto"`. + +## The `auto` rule + +`auto` resolves to `mineru` when the `mineru` CLI is on `PATH`, otherwise +to `pymupdf`. mineru has given the best results and remains the reference +backend wherever it's installed — `auto` never prefers the lighter +backend over an installed mineru. The light backend exists to make +*installing* mineru optional, not to replace it where it's present. + +| | `mineru` | `pymupdf` | +|--------------------|-----------------------------------|-------------------------------------| +| Quality | Reference: the best structure document2md produces | Slightly worse structure | +| Speed | Slow: runs real layout/OCR models | Orders of magnitude faster | +| Install size | Heavy (`mineru[pipeline]`, an optional extra, plus system OpenCV packages) | Small, core dependency | +| What it can read | PDFs and scanned page images, digital or scanned alike | Only a PDF with enough of an embedded text layer | +| Figures | Extracted, alongside the Markdown | Not extracted | + +## What counts as "enough" of a text layer + +The `pymupdf` backend can't tell the difference between a scanned PDF and +one it simply doesn't understand, so it decides up front, page by page: +a page counts as text-bearing when its extracted text has at least 50 +non-whitespace characters, and the PDF as a whole qualifies when at least +80% of its pages are text-bearing — a whole-document decision, tolerating +a handful of scanned inserts in an otherwise born-digital document. A +fully scanned PDF always fails this check. + +A list of scanned page images always needs OCR — under `pymupdf` it's +rejected outright, the same as a PDF that fails the text-layer check. +Neither case falls back silently to `mineru`: document2md raises an error +naming `pip install "document2md[mineru]"` instead, so the choice stays +visible rather than hidden behind a slow, surprising OCR pass. See +[Command line](cli.qmd#what-the-error-messages-mean) for the exact wording. + +## Forcing a backend + +- **Force `mineru`** (`--backend mineru`) when you want the best + structure, or a scanned document, and don't mind the wait or the + install size. +- **Force `pymupdf`** (`--backend pymupdf`) when you know every document + in a batch is born-digital and want to fail fast (`RuntimeError`, + not a long OCR pass) on the rare one that turns out not to be. +- **Leave it at `auto`** (the default) otherwise: it picks the best + backend actually available, without forcing an install either way. + +## How the output differs + +Both backends produce Markdown; the two differ in what they can capture +from the source PDF: + +- **Headings.** `mineru` infers structure from its layout model; + `pymupdf` infers headings from font sizes. +- **Tables.** `mineru`'s complex tables (rowspan/colspan) arrive as raw + HTML and are rewritten into Markdown tables by document2md itself (see + the developer docs' architecture page); `pymupdf4llm` emits Markdown + tables directly via PyMuPDF's own table finder. +- **Figures.** `mineru` extracts figures alongside the Markdown, into + `_images/`; `pymupdf` extracts none. diff --git a/website/pages/cli.qmd b/website/pages/cli.qmd new file mode 100644 index 0000000..d1d489b --- /dev/null +++ b/website/pages/cli.qmd @@ -0,0 +1,90 @@ +--- +title: "Command line" +--- + +`document2md` takes exactly one input source — a local PDF or a set of +local page images — and converts it to Markdown. It never downloads +anything itself; get the file first, then convert it. + +## Input + +```bash +document2md --pdf edicion.pdf +``` + +`--pdf` converts a single local PDF file. + +```bash +document2md --images pagina-1.jpg pagina-2.jpg --filename out.md +``` + +`--images` converts an ordered list of scanned page image files — one note +spanning several pages. `--filename` sets the output Markdown's name; with +`--pdf` it defaults to the PDF's own name (`edicion.pdf` → `edicion.md`), +but with `--images` it's required, since a set of images has no single +name to derive one from. `--outdir` sets the output directory (default: +`output/`). + +## Choosing a backend + +```bash +document2md --pdf edicion.pdf --backend mineru +``` + +`--backend {auto,mineru,pymupdf}` (default `auto`) selects which backend +converts the document. `auto` resolves to `mineru` when it's on `PATH`, +otherwise `pymupdf`; see [Backends](backends.qmd) for the full rule and +what each backend can and can't handle. + +## Cropping to one note + +```bash +document2md --pdf edicion.pdf \ + --titulo "ACUERDO por el que se..." \ + --titulo-siguiente "DECRETO por el que se..." +``` + +Since one edition's PDF holds every note published that day, +`--titulo`/`--titulo-siguiente` crop the resulting Markdown down to just +one note — its own title, and the next note's title, as they appear in the +gazette's own index. Title matching is fuzzy (OCR text rarely matches an +index title exactly), so a match below `--min-confidence` (default `0.6`) +is treated as not found and the crop falls back to keeping more text +rather than dropping content. + +Other flags: + +- `--keep-pages` — also keep the uncropped Markdown, as + `/.full.md`. +- `--keep-mineru-output` — keep mineru's own raw output (layout/model + JSON, rendered PDFs...) in `/_mineru/` instead of + discarding it; useful when a conversion looks wrong and mineru's own + read of the page is the first thing worth inspecting. Ignored under the + `pymupdf` backend, which produces no such output. + +## What the error messages mean + +document2md exits with a message and no traceback, rather than a stack +trace, whenever it can't proceed: + +- **`Provide exactly one of: --pdf or --images.`** — neither, or both, + were given. +- **`--filename is required with --images (there's no single input name to + derive one from).`** +- **`Conversion timed out after 3600s. This document may be unusually + large.`** — the `mineru` backend didn't finish in time. +- **`'mineru' is required for the mineru backend but isn't installed. + Install it with: pip install "document2md[mineru]"`** — `--backend + mineru` (or `auto` resolving to it) was requested without the `mineru` + extra installed. +- **`Scanned page images always need OCR. Install the mineru backend with: + pip install "document2md[mineru]"`** — `--images` was combined with + `--backend pymupdf`. +- **` has no embedded text layer and needs OCR. Install the mineru + backend with: pip install "document2md[mineru]"`** — the `pymupdf` + backend was asked to convert a scanned (or mostly scanned) PDF; see + [Backends](backends.qmd) for exactly what "has no embedded text layer" + means. + +None of these fall back silently to another backend — document2md tells +you what to install or what to change instead. diff --git a/website/pages/install.qmd b/website/pages/install.qmd new file mode 100644 index 0000000..ea7bd92 --- /dev/null +++ b/website/pages/install.qmd @@ -0,0 +1,53 @@ +--- +title: "Install" +--- + +document2md needs Python 3.10 or later. + +```bash +pip install document2md +``` + +installs the light install: the `pymupdf` backend, built on +[pymupdf4llm](https://github.com/pymupdf/RAG), which reads a born-digital +PDF's own embedded text layer. No OCR, no models, small wheels. It cannot +convert a scanned PDF or a set of scanned page images — see +[Backends](backends.qmd) for what counts as "born-digital enough". + +```bash +pip install "document2md[mineru]" +``` + +adds the `mineru` backend, for OCR and layout analysis on scanned pages. +`mineru[pipeline]` is a heavy dependency: expect a large download, and a +one-time model download the first time it actually runs. On Debian/Ubuntu, +mineru's OpenCV dependency also needs two system packages: + +```bash +sudo apt-get install libgl1 libglib2.0-0 +``` + +**License note.** PyMuPDF and pymupdf4llm — the light `pymupdf` backend's +dependency — are AGPL-3.0 licensed, unlike document2md itself +(Apache-2.0). Check that fits your project before redistributing. + +## Which backend do I have? + +```bash +document2md --help +``` + +always lists `--backend {auto,mineru,pymupdf}`, whether or not `mineru` is +installed. To find out which one `auto` actually resolves to on your +machine, run a conversion and read the first line it prints: + +```console +$ document2md --pdf edicion.pdf +Converting to Markdown (pymupdf)... +``` + +If you ask for a backend that isn't installed, or for `pymupdf` on input it +can't handle, document2md exits with a message naming exactly what to +install — never a traceback. See [Backends](backends.qmd) for the full +`auto` rule, and [Command line](cli.qmd#what-the-error-messages-mean) for +the exact wording of every such message. diff --git a/website/pages/python.qmd b/website/pages/python.qmd new file mode 100644 index 0000000..d9246be --- /dev/null +++ b/website/pages/python.qmd @@ -0,0 +1,99 @@ +--- +title: "Python" +--- + +`BatchConverter` is document2md's Python entry point — the same code the +`document2md` command runs, usable directly from a script. The code blocks +on this page are illustrative and are not executed when the site is built; +see [Command line](cli.qmd) for a version you can copy and run as-is, or +the developer docs for a fully worked, tested example. + +## Converting a batch of documents + +```python +from document2md import BatchConverter + +jobs = [ + ("a.pdf", "output", "a.md"), + (["b-p1.jpg", "b-p2.jpg"], "output", "b.md"), +] + +with BatchConverter() as convert: + for path_or_paths, outdir, filename in jobs: + convert(path_or_paths, outdir, filename) +``` + +`BatchConverter()` (default `backend="auto"`) is a context manager: +entering it resolves the backend and, only for `mineru`, starts a +persistent `mineru-api` server that every conversion in the `with` block +shares — worth it once a batch has more than a document or two, since it +avoids reloading mineru's models per document. Exiting stops that server. +Each call to the entered converter takes a single PDF path, or a list of +image paths for a document spanning several scanned pages, and writes the +result to `outdir/filename`. + +## Choosing a backend + +```python +with BatchConverter(backend="mineru") as convert: + ... +``` + +`backend` accepts `"auto"` (default), `"mineru"` or `"pymupdf"`; anything +else raises `ValueError`. Once entered, the resolved name is available as +`convert.backend`. See [Backends](backends.qmd) for the `auto` rule. + +## Every option of a single conversion + +```python +with BatchConverter() as convert: + convert( + "edicion.pdf", "output", "nota.md", + titulo="ACUERDO por el que se...", + titulo_siguiente="DECRETO por el que se...", + min_confidence=0.8, + keep_pages=True, + keep_mineru_output=True, + timeout=1800, + ) +``` + +- `titulo`/`titulo_siguiente` — crop the result down to the text between + the two titles, as they appear in the gazette's own index. Left out + (the default), the whole conversion is kept as-is. +- `min_confidence` (default `0.6`) — how confident a fuzzy title match has + to be before it's trusted; a weaker one falls back to keeping more text. +- `keep_pages` — also keep the uncropped Markdown, as + `/.full.md`. +- `keep_mineru_output`, `timeout` — forwarded to the `mineru` backend; + accepted but ignored under `pymupdf`, which needs neither. + +## Sharing a server with a caller further up + +```python +import os +os.environ["MINERU_API_URL"] # already set by a caller further up the stack +with BatchConverter() as convert: + ... # skips starting its own mineru-api server +``` + +If `MINERU_API_URL` is already set when `BatchConverter` is entered — set +by a caller further up that is running its own batch — it reuses that +server instead of starting a second one. + +## Used from LegalIA's nota2md + +`nota2md.legal_provisions` accepts an already-`__enter__`'d +`BatchConverter` as its own `converter` parameter, so a batch of DOF legal +provisions can share the same warm server too — the way to OCR many +pre-HTML-era notes without paying mineru's startup cost once per note (see +[`nota2md` in the LegalIA +repository](https://github.com/INGEOTEC/LegalIA/tree/master/packages/nota2md)): + +```python +import nota2md + +with BatchConverter() as convert: + for cod_nota in codigos_sin_html: + nota2md.legal_provisions(cod_nota, "output", source="image", converter=convert) +``` diff --git a/website/style.scss b/website/style.scss new file mode 100644 index 0000000..5b78c21 --- /dev/null +++ b/website/style.scss @@ -0,0 +1,149 @@ +/*-- scss:defaults --*/ + +// Academic look: serif text, narrow reading measure, restrained color. +$font-family-base: "Palatino Linotype", Palatino, "URW Palladio L", Georgia, + "Times New Roman", serif; +$headings-font-family: $font-family-base; +$font-size-root: 19px; +$line-height-base: 1.65; + +$body-bg: #fdfdfb; +$body-color: #1f1e1b; +$link-color: #1c5cab; +$link-decoration: none; + +$grid-body-width: 760px; +$grid-sidebar-width: 260px; +$grid-margin-width: 260px; + +/*-- scss:rules --*/ + +main p { + text-align: justify; + hyphens: auto; +} + +h1, +h2, +h3 { + font-weight: 600; +} + +h1.title { + font-size: 2rem; +} + +.subtitle.lead { + font-style: italic; + color: #52514e; +} + +// The abstract as an academic block. +.abstract { + margin: 1.5rem 0; + padding: 0 1.5rem; + border-left: 3px solid #e1e0d9; + font-size: 0.92em; + + .abstract-title { + font-variant: small-caps; + letter-spacing: 0.05em; + } +} + +.figure-caption, +.table-caption, +caption { + font-size: 0.85em; + color: #52514e; + text-align: left; +} + +.table { + font-size: 0.85em; + font-family: system-ui, -apple-system, "Segoe UI", sans-serif; +} + +// Figures inside tables align vertically. +.table td { + font-variant-numeric: tabular-nums; +} + +// Landing page. +.home-lead p { + font-size: 1.22em; + line-height: 1.5; + text-align: left; + hyphens: none; + margin: 1.6rem 0 2rem; +} + +.fronts { + display: grid; + grid-template-columns: repeat(3, 1fr); + gap: 0.9rem; + margin: 2.2rem 0 2.6rem; + + .front { + border: 1px solid #e1e0d9; + border-top: 3px solid #1c5cab; + background: #f9f8f4; + padding: 1rem 1.1rem 1.1rem; + + h3 { + margin: 0 0 0.5rem; + font-size: 0.78rem; + font-family: system-ui, -apple-system, "Segoe UI", sans-serif; + font-weight: 600; + text-transform: uppercase; + letter-spacing: 0.08em; + color: #1c5cab; + } + + p { + margin: 0; + font-size: 0.85em; + line-height: 1.5; + text-align: left; + hyphens: none; + color: #3c3b37; + } + } +} + +.team { + display: grid; + grid-template-columns: repeat(2, 1fr); + gap: 0.4rem 1.4rem; + margin: 1.4rem 0 1.8rem; + + .member p { + margin: 0; + text-align: left; + line-height: 1.45; + border-left: 3px solid #e1e0d9; + padding-left: 0.9rem; + + strong { + font-weight: 600; + } + } +} + +@media (max-width: 768px) { + .fronts, + .team { + grid-template-columns: 1fr; + } +} + +.navbar { + border-bottom: 1px solid #e1e0d9; + font-family: system-ui, -apple-system, "Segoe UI", sans-serif; + font-size: 0.85rem; +} + +.page-footer { + font-size: 0.8rem; + color: #898781; +}