Repository navigation
AP-859 pytesseract worker update #5
New issue
Have a question about this project? Sign up for a free GitHub account to open an issue and contact its maintainers and the community.
By clicking “Sign up for GitHub”, you agree to our terms of service and privacy statement. We’ll occasionally send you account related emails.
Already on GitHub? Sign in to your account
Changes from all commits
File filter
Filter by extension
Conversations
Jump to
Diff view
Diff view
There are no files selected for viewing
| Original file line number | Diff line number | Diff line change |
|---|---|---|
|
|
@@ -220,3 +220,5 @@ __marimo__/ | |
| # other stuff | ||
| artifacts/* | ||
| uv.lock | ||
| files/* | ||
| .DS_Store | ||
| Original file line number | Diff line number | Diff line change |
|---|---|---|
|
|
@@ -12,6 +12,7 @@ dependencies = [ | |
| "flower", | ||
| "gunicorn", | ||
| "psycopg[c]", | ||
| "pytesseract", | ||
| "redis", | ||
| "sqlalchemy" | ||
| ] | ||
|
|
||
| Original file line number | Diff line number | Diff line change |
|---|---|---|
| @@ -1,8 +1,11 @@ | ||
| """Celery tasks for running OCR jobs.""" | ||
|
|
||
| import subprocess | ||
| import hashlib | ||
| from pathlib import Path | ||
|
|
||
| # directly imports run_tesseract which is not explicitly exported by the pytesseract package | ||
| # The two exported functions we could use force a tmp file to be created and then deleted. | ||
| from pytesseract.pytesseract import run_tesseract | ||
|
anarchivist marked this conversation as resolved.
|
||
| from celery import shared_task | ||
|
|
||
|
|
||
|
|
@@ -22,7 +25,19 @@ def run_tesseract_job(self, filelist: str, languages: list[str], output: str) -> | |
| meta={"filelist": filelist, "languages": languages, "output": output}, | ||
| ) | ||
|
|
||
| command = ["tesseract", "-l", "+".join(languages), filelist, output, "pdf"] | ||
| subprocess.run(command, check=True, capture_output=True, text=True) | ||
|
|
||
| return {"output": f"{output}.pdf"} | ||
| kwargs = { | ||
| "input_filename": filelist, | ||
| "output_filename_base": output, | ||
| "extension": "pdf", | ||
| "lang": "+".join(languages), | ||
| } | ||
|
|
||
| try: | ||
| run_tesseract(**kwargs) | ||
|
Member
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. I assume that we are avoiding Just want to make sure that I'm understanding the rationale for using a private/undocumented API.
Contributor
Author
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. You are correct and it is possibly a little brittle because of it.
Member
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. yep, that's correct - the there was some discussion on the potential signature of |
||
| output_path = Path(f"{output}.pdf") | ||
| with output_path.open("rb") as f: | ||
| sha256 = hashlib.file_digest(f, "sha256").hexdigest() | ||
| except Exception as e: | ||
| raise RuntimeError(f"Error running Tesseract: {e}") from e | ||
|
|
||
| return {"output_path": str(output_path), "sha256": sha256} | ||
Uh oh!
There was an error while loading. Please reload this page.