Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -220,3 +220,5 @@ __marimo__/
# other stuff
artifacts/*
uv.lock
files/*
.DS_Store
11 changes: 6 additions & 5 deletions docker-compose.yml
Original file line number Diff line number Diff line change
Expand Up @@ -46,8 +46,8 @@ services:
ports:
- 8000:8000
volumes:
- ./quiabo:/app/quiabo:rw
- ./test:/app/test:rw
- ./:/app
- ./files:/srv/files:ro

worker:
build:
Expand All @@ -66,12 +66,13 @@ services:
restart: always
command: celery -A quiabo.celery_app worker --loglevel INFO
volumes:
- ./quiabo:/app/quiabo:rw
- ./files:/app/files:rw
- ./:/app
- ./files:/srv/files

flower:
build:
context: .
target: app
profiles:
- flower
depends_on:
Expand All @@ -91,7 +92,7 @@ services:
ports:
- 127.0.0.1:5555:5555
volumes:
Comment thread
anarchivist marked this conversation as resolved.
- ./quiabo:/app/quiabo:rw
- ./:/app

redis:
healthcheck:
Expand Down
1 change: 1 addition & 0 deletions pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -12,6 +12,7 @@ dependencies = [
"flower",
"gunicorn",
"psycopg[c]",
"pytesseract",
"redis",
"sqlalchemy"
]
Expand Down
25 changes: 20 additions & 5 deletions quiabo/tasks.py
Original file line number Diff line number Diff line change
@@ -1,8 +1,11 @@
"""Celery tasks for running OCR jobs."""

import subprocess
import hashlib
from pathlib import Path

# directly imports run_tesseract which is not explicitly exported by the pytesseract package
# The two exported functions we could use force a tmp file to be created and then deleted.
from pytesseract.pytesseract import run_tesseract
Comment thread
anarchivist marked this conversation as resolved.
from celery import shared_task


Expand All @@ -22,7 +25,19 @@ def run_tesseract_job(self, filelist: str, languages: list[str], output: str) ->
meta={"filelist": filelist, "languages": languages, "output": output},
)

command = ["tesseract", "-l", "+".join(languages), filelist, output, "pdf"]
subprocess.run(command, check=True, capture_output=True, text=True)

return {"output": f"{output}.pdf"}
kwargs = {
"input_filename": filelist,
"output_filename_base": output,
"extension": "pdf",
"lang": "+".join(languages),
}

try:
run_tesseract(**kwargs)

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I assume that we are avoiding run_and_get_output because it stores the result in memory instead of on disk. And I assume image_to_pdf_or_hocr doesn't allow us to customise the output path?

Just want to make sure that I'm understanding the rationale for using a private/undocumented API.

Copy link
Copy Markdown
Contributor Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

You are correct and it is possibly a little brittle because of it.

@anarchivist anarchivist Sep 30, 2026 •

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

yep, that's correct - the return_bytes parameter that gets set basically means "should I return bytes or str?"

there was some discussion on the potential signature of run_tesseract as "possibly changing year to year" in an issue on the pytesseract repository, but a maintainer made that comment in 2018. the signature hasn't changed in 7 years.

output_path = Path(f"{output}.pdf")
with output_path.open("rb") as f:
sha256 = hashlib.file_digest(f, "sha256").hexdigest()
except Exception as e:
raise RuntimeError(f"Error running Tesseract: {e}") from e

return {"output_path": str(output_path), "sha256": sha256}
92 changes: 71 additions & 21 deletions test/unit/test_tasks.py
Original file line number Diff line number Diff line change
@@ -1,7 +1,11 @@
"""Unit tests for the shared task defined in ``tasks.py``."""

import hashlib
from pathlib import Path
from unittest.mock import Mock

import pytest

from quiabo.tasks import run_tesseract_job


Expand All @@ -28,57 +32,103 @@ def test_shared_task_delay_delegates_to_apply_async(monkeypatch):
)


def test_run_tesseract_job_updates_state_and_runs_tesseract(monkeypatch):
"""Verify that the task updates state and invokes Tesseract."""
def test_run_tesseract_job_updates_state_and_returns_digest(monkeypatch, tmp_path):
"""Verify that the task runs Tesseract and returns the PDF digest."""
filelist = tmp_path / "files.txt"
filelist.touch()
output = tmp_path / "output"
pdf_contents = b"generated PDF"

update_state = Mock()
run = Mock()

def write_output(**kwargs):
Path(f"{kwargs['output_filename_base']}.pdf").write_bytes(pdf_contents)

run.side_effect = write_output
monkeypatch.setattr(run_tesseract_job, "update_state", update_state)
monkeypatch.setattr("quiabo.tasks.subprocess.run", run)
monkeypatch.setattr("quiabo.tasks.run_tesseract", run)

result = run_tesseract_job.run(
"files.txt",
str(filelist),
["eng", "spa"],
"output",
str(output),
)

update_state.assert_called_once_with(
state="STARTED",
meta={
"filelist": "files.txt",
"filelist": str(filelist),
"languages": ["eng", "spa"],
"output": "output",
"output": str(output),
},
)
run.assert_called_once_with(
["tesseract", "-l", "eng+spa", "files.txt", "output", "pdf"],
check=True,
capture_output=True,
text=True,
input_filename=str(filelist),
output_filename_base=str(output),
extension="pdf",
lang="eng+spa",
)
assert result == {"output": "output.pdf"}
assert result == {
"output_path": f"{output}.pdf",
"sha256": hashlib.sha256(pdf_contents).hexdigest(),
}


def test_run_tesseract_job_removes_pdf_suffix(monkeypatch):
def test_run_tesseract_job_removes_pdf_suffix(monkeypatch, tmp_path):
"""Verify that an existing PDF suffix is removed before invocation."""
update_state = Mock()
run = Mock()
output = tmp_path / "output.pdf"
output_base = str(output.with_suffix(""))

def write_output(**kwargs):
Path(f"{kwargs['output_filename_base']}.pdf").touch()

run.side_effect = write_output
monkeypatch.setattr(run_tesseract_job, "update_state", update_state)
monkeypatch.setattr("quiabo.tasks.subprocess.run", run)
monkeypatch.setattr("quiabo.tasks.run_tesseract", run)

result = run_tesseract_job.run("files.txt", ["eng"], "output.pdf")
result = run_tesseract_job.run("files.txt", ["eng"], str(output))

update_state.assert_called_once_with(
state="STARTED",
meta={
"filelist": "files.txt",
"languages": ["eng"],
"output": "output",
"output": output_base,
},
)
run.assert_called_once_with(
["tesseract", "-l", "eng", "files.txt", "output", "pdf"],
check=True,
capture_output=True,
text=True,
input_filename="files.txt",
output_filename_base=output_base,
extension="pdf",
lang="eng",
)
assert result == {"output": "output.pdf"}
assert result == {
"output_path": str(output),
"sha256": hashlib.sha256(b"").hexdigest(),
}


def test_run_tesseract_job_requires_existing_output_directory(tmp_path):
"""Verify that the task rejects a missing output directory."""
output = tmp_path / "missing" / "output"

with pytest.raises(FileNotFoundError, match="does not exist"):
run_tesseract_job.run("files.txt", ["eng"], str(output))


def test_run_tesseract_job_wraps_tesseract_errors(monkeypatch, tmp_path):
"""Verify that Tesseract errors are wrapped with task context."""
output = tmp_path / "output"
error = RuntimeError("Tesseract failed")
run = Mock(side_effect=error)
monkeypatch.setattr("quiabo.tasks.run_tesseract", run)

with pytest.raises(
RuntimeError, match="Error running Tesseract: Tesseract failed"
) as exc_info:
run_tesseract_job.run("files.txt", ["eng"], str(output))

assert exc_info.value.__cause__ is error
Loading