From c5d3ef4b17f80eb1bc9c13ef2cee3a8a3fc66b80 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Tue, 27 Jan 2026 14:04:52 -0800 Subject: [PATCH] Tighten ruff rules and modernize style --- docs/conf.py | 5 +- misc/_webservice.py | 80 ++++++++++++++------------ misc/batch.py | 5 +- misc/bisect_pdf.py | 1 + misc/watcher.py | 13 ++--- misc/webservice.py | 4 +- pyproject.toml | 24 +++++--- src/ocrmypdf/_defaults.py | 2 + src/ocrmypdf/_exec/ghostscript.py | 9 +++ src/ocrmypdf/_metadata.py | 4 +- src/ocrmypdf/_pipelines/ocr.py | 2 +- src/ocrmypdf/font/__init__.py | 1 + src/ocrmypdf/fpdf_renderer/__init__.py | 1 + src/ocrmypdf/helpers.py | 1 - src/ocrmypdf/hocrtransform/__main__.py | 1 + src/ocrmypdf/languages.py | 1 + tests/conftest.py | 7 ++- tests/test_json_serialization.py | 1 + tests/test_metadata.py | 8 +-- tests/test_multilingual_direct.py | 1 + tests/test_unpaper.py | 2 +- tests/test_watcher.py | 8 ++- 22 files changed, 104 insertions(+), 77 deletions(-) diff --git a/docs/conf.py b/docs/conf.py index 96927133..a4a8f636 100755 --- a/docs/conf.py +++ b/docs/conf.py @@ -25,10 +25,11 @@ # sys.path.insert(0, os.path.abspath('.')) # -- General configuration ------------------------------------------------ +from __future__ import annotations needs_sphinx = '8' -import datetime +import datetime as dt # Add any Sphinx extension module names here, as strings. They can be # extensions coming with Sphinx (named 'sphinx.ext.*') or your custom @@ -63,7 +64,7 @@ master_doc = 'index' # General information about the project. project = 'ocrmypdf' -year = str(datetime.date.today().year) +year = str(dt.date.today().year) copyright = ( f'{year}, James R. Barlow. ' + 'Licensed under Creative Commons Attribution-ShareAlike 4.0' diff --git a/misc/_webservice.py b/misc/_webservice.py index 6016080e..df7752f0 100644 --- a/misc/_webservice.py +++ b/misc/_webservice.py @@ -96,7 +96,9 @@ with st.expander("Optimization after OCR"): png_quality = st.slider( "PNG quality", min_value=0, max_value=100, value=75, key="png_quality" ) - jbig2_threshold = st.number_input("JBIG2 threshold", value=0.85, key="jbig2_threshold") + jbig2_threshold = st.number_input( + "JBIG2 threshold", value=0.85, key="jbig2_threshold" + ) with st.expander("Advanced options"): jobs = st.slider( @@ -192,45 +194,47 @@ if uploaded: args.append(f"--jbig2-threshold={jbig2_threshold}") if jobs: args.append(f"--jobs={jobs}") - input_file = NamedTemporaryFile(delete=True, suffix=f"_{uploaded.name}") - input_file.write(uploaded.getvalue()) - input_file.flush() - input_file.seek(0) - args.append(str(input_file.name)) - output_file = NamedTemporaryFile(delete=True, suffix=".pdf") - args.append(str(output_file.name)) + with NamedTemporaryFile(delete=True, suffix=f"_{uploaded.name}") as input_file: + input_file.write(uploaded.getvalue()) + input_file.flush() + input_file.seek(0) + args.append(str(input_file.name)) + with NamedTemporaryFile(delete=True, suffix=".pdf") as output_file: + args.append(str(output_file.name)) - st.session_state['running'] = ( - 'run_button' in st.session_state and st.session_state.run_button - ) - if st.button( - "Run OCRmyPDF", - disabled=st.session_state.get("running", False), - key='run_button', - ): - st.session_state['running'] = True - args = [sys.executable, '-u', '-m', "ocrmypdf"] + args + st.session_state['running'] = ( + 'run_button' in st.session_state and st.session_state.run_button + ) + if st.button( + "Run OCRmyPDF", + disabled=st.session_state.get("running", False), + key='run_button', + ): + st.session_state['running'] = True + args = [sys.executable, '-u', '-m', "ocrmypdf"] + args - proc = subprocess.Popen(args, stdout=subprocess.PIPE, stderr=subprocess.PIPE) - with st.container(border=True): - while proc.poll() is None: - line = proc.stderr.readline() - if line: - st.html("" + line.decode().strip() + "") + proc = subprocess.Popen( + args, stdout=subprocess.PIPE, stderr=subprocess.PIPE + ) + with st.container(border=True): + while proc.poll() is None: + line = proc.stderr.readline() + if line: + st.html("" + line.decode().strip() + "") - if proc.returncode != 0: - st.error(f"ocrmypdf failed with exit code {proc.returncode}") - st.session_state['running'] = False - st.stop() + if proc.returncode != 0: + st.error(f"ocrmypdf failed with exit code {proc.returncode}") + st.session_state['running'] = False + st.stop() - if Path(output_file.name).stat().st_size == 0: - st.error("No output PDF file was generated") - st.stop() + if Path(output_file.name).stat().st_size == 0: + st.error("No output PDF file was generated") + st.stop() - st.download_button( - label="Download output PDF", - data=output_file.read(), - file_name=uploaded.name, - mime="application/pdf", - ) - st.session_state['running'] = False + st.download_button( + label="Download output PDF", + data=output_file.read(), + file_name=uploaded.name, + mime="application/pdf", + ) + st.session_state['running'] = False diff --git a/misc/batch.py b/misc/batch.py index a9e45ac2..f45e8919 100644 --- a/misc/batch.py +++ b/misc/batch.py @@ -39,10 +39,7 @@ script_dir = Path(__file__).parent # set archive_dir to a path for backup original documents. Leave empty if not required. archive_dir = "/pdfbak" -if len(sys.argv) > 1: - start_dir = Path(sys.argv[1]) -else: - start_dir = Path(".") +start_dir = Path(sys.argv[1]) if len(sys.argv) > 1 else Path(".") if len(sys.argv) > 2: log_file = Path(sys.argv[2]) diff --git a/misc/bisect_pdf.py b/misc/bisect_pdf.py index 79cd8f98..b50aec07 100644 --- a/misc/bisect_pdf.py +++ b/misc/bisect_pdf.py @@ -3,6 +3,7 @@ # SPDX-License-Identifier: MIT """Helper script for bisecting PDFs to find a page with an issue.""" +from __future__ import annotations import sys diff --git a/misc/watcher.py b/misc/watcher.py index 95447905..7d1f421f 100644 --- a/misc/watcher.py +++ b/misc/watcher.py @@ -7,12 +7,12 @@ from __future__ import annotations +import datetime as dt import json import logging import shutil import sys import time -from datetime import datetime from enum import Enum from pathlib import Path from typing import Annotated, Any @@ -48,7 +48,7 @@ class LoggingLevelEnum(str, Enum): def get_output_path(root: Path, basename: str, output_dir_year_month: bool) -> Path: assert '/' not in basename, "basename must not contain '/'" if output_dir_year_month: - today = datetime.today() + today = dt.datetime.today() output_directory_year_month = root / str(today.year) / f'{today.month:02d}' if not output_directory_year_month.exists(): output_directory_year_month.mkdir(parents=True, exist_ok=True) @@ -140,7 +140,7 @@ class HandleObserverEvent(PatternMatchingEventHandler): ignore_patterns=None, ignore_directories=False, case_sensitive=False, - settings={}, + settings=None, ): super().__init__( patterns=patterns, @@ -148,7 +148,7 @@ class HandleObserverEvent(PatternMatchingEventHandler): ignore_directories=ignore_directories, case_sensitive=case_sensitive, ) - self._settings = settings + self._settings = settings if settings else {} def on_any_event(self, event): if event.event_type in ['created']: @@ -302,10 +302,7 @@ def main( 'output_dir_year_month': output_dir_year_month, }, ) - if use_polling: - observer = PollingObserver() - else: - observer = Observer() + observer = PollingObserver() if use_polling else Observer() observer.schedule(handler, input_dir, recursive=True) observer.start() print(f"Watching {input_dir} for new PDFs. Press Ctrl+C to exit.") diff --git a/misc/webservice.py b/misc/webservice.py index a432f50e..d3584414 100755 --- a/misc/webservice.py +++ b/misc/webservice.py @@ -4,6 +4,8 @@ """Run the OCRmyPDF web service.""" +from __future__ import annotations + import os import sys @@ -13,7 +15,7 @@ except ImportError: raise ImportError( 'You need to install streamlit in the Python environment ' 'to run the web service.\n' - ) + ) from None if __name__ == '__main__': os.execvp( diff --git a/pyproject.toml b/pyproject.toml index 06da8f66..01d95e9b 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -130,6 +130,7 @@ exclude = ["src/ocrmypdf/_version.py"] # Autogenerated "UP", # pyupgrade "SIM", # simplify "B", # flake8-bugbear + "ICN", # flake8-import-conventions ] ignore = [ "B028", # warning with no explicit stacklevel @@ -139,13 +140,20 @@ ignore = [ [tool.ruff.lint.isort] known-first-party = ["ocrmypdf"] +required-imports = ["from __future__ import annotations"] + +[tool.ruff.lint.flake8-import-conventions] +# Prohibit explicit imports from the 'datetime' module +banned-from = ["datetime"] +# Optionally, suggest an alias for 'import datetime' (e.g., as dt) +extend-aliases = { "datetime" = "dt" } [tool.ruff.lint.pydocstyle] convention = "google" [tool.ruff.lint.per-file-ignores] "docs/conf.py" = ["D100", "D101", "D105"] -"tests/*.py" = ["D100", "D101", "D102", "D103", "D105"] +"tests/*.py" = ["D100", "D101", "D102", "D103", "D105", "E501"] "misc/*.py" = ["D103", "D101", "D102"] "src/ocrmypdf/builtin_plugins/*.py" = ["D103", "D102", "D105"] @@ -154,11 +162,7 @@ quote-style = "preserve" [dependency-groups] # Developer-only tools - use `uv sync --group ` -dev = [ - "mypy>=1.13.0", - "ipykernel>=6.29.5", - "reportlab>=4.4.4", -] +dev = ["mypy>=1.13.0", "ipykernel>=6.29.5", "reportlab>=4.4.4"] test = [ # Core testing framework "coverage[toml]>=6.2", @@ -175,5 +179,11 @@ test = [ # Extended test capabilities (merged from extended_test) "pymupdf>=1.24.14", ] -docs = ["myst-parser>=4.0.1", "sphinx", "sphinx-issues", "sphinx-rtd-theme", "sphinxcontrib-mermaid"] +docs = [ + "myst-parser>=4.0.1", + "sphinx", + "sphinx-issues", + "sphinx-rtd-theme", + "sphinxcontrib-mermaid", +] streamlit-dev = ["streamlit>=1.40.2", "streamlit-pdf-viewer>=0.0.19"] diff --git a/src/ocrmypdf/_defaults.py b/src/ocrmypdf/_defaults.py index 39674286..c979b8eb 100644 --- a/src/ocrmypdf/_defaults.py +++ b/src/ocrmypdf/_defaults.py @@ -2,6 +2,8 @@ # SPDX-License-Identifier: MPL-2.0 # Enforce English hegemony +from __future__ import annotations + DEFAULT_LANGUAGE = 'eng' # Default rotation threshold diff --git a/src/ocrmypdf/_exec/ghostscript.py b/src/ocrmypdf/_exec/ghostscript.py index 92f757fe..b48f2080 100644 --- a/src/ocrmypdf/_exec/ghostscript.py +++ b/src/ocrmypdf/_exec/ghostscript.py @@ -111,6 +111,15 @@ def rasterize_pdf( """Rasterize one page of a PDF at resolution raster_dpi in canvas units. Args: + input_file: The PDF file to rasterize. + output_file: The file to write the rasterized PDF to. + raster_device: The Ghostscript raster device to use to rasterize the PDF. + raster_dpi: Resolution in dots per inch at which to rasterize page. + pageno: Page number to rasterize (beginning at page 1). + page_dpi: Resolution, overriding output image DPI. + rotation: Cardinal angle, clockwise, to rotate page. + filter_vector: If True, remove vector graphics objects. + stop_on_error: If True, stop rasterizing on the first error. use_cropbox: If True, rasterize the CropBox instead of MediaBox. Default is False (use MediaBox). """ diff --git a/src/ocrmypdf/_metadata.py b/src/ocrmypdf/_metadata.py index 74c13022..f973a2a0 100644 --- a/src/ocrmypdf/_metadata.py +++ b/src/ocrmypdf/_metadata.py @@ -5,9 +5,9 @@ from __future__ import annotations +import datetime as dt import logging import os -from datetime import datetime, timezone from pathlib import Path from typing import Any @@ -53,7 +53,7 @@ def get_docinfo(base_pdf: Pdf, context: PdfContext) -> dict[str, str]: pdfmark['/Creator'] = f'{PROGRAM_NAME} {OCRMYPF_VERSION} / {creator_tag}' pdfmark['/Producer'] = f'pikepdf {PIKEPDF_VERSION}' - pdfmark['/ModDate'] = encode_pdf_date(datetime.now(timezone.utc)) + pdfmark['/ModDate'] = encode_pdf_date(dt.datetime.now(dt.UTC)) return pdfmark diff --git a/src/ocrmypdf/_pipelines/ocr.py b/src/ocrmypdf/_pipelines/ocr.py index d875be0e..dd0288e7 100644 --- a/src/ocrmypdf/_pipelines/ocr.py +++ b/src/ocrmypdf/_pipelines/ocr.py @@ -50,7 +50,7 @@ from ocrmypdf._validation import ( ) from ocrmypdf.exceptions import ExitCode from ocrmypdf.helpers import available_cpu_count -from ocrmypdf.hocrtransform.ocr_element import OcrElement +from ocrmypdf.models.ocr_element import OcrElement log = logging.getLogger(__name__) diff --git a/src/ocrmypdf/font/__init__.py b/src/ocrmypdf/font/__init__.py index 306808d7..f94cc3cb 100644 --- a/src/ocrmypdf/font/__init__.py +++ b/src/ocrmypdf/font/__init__.py @@ -10,6 +10,7 @@ This module provides font infrastructure for the fpdf2 PDF renderer. It includes - MultiFontManager: Automatic font selection for multilingual documents - SystemFontProvider: System font discovery """ +from __future__ import annotations from ocrmypdf.font.font_manager import FontManager from ocrmypdf.font.font_provider import ( diff --git a/src/ocrmypdf/fpdf_renderer/__init__.py b/src/ocrmypdf/fpdf_renderer/__init__.py index 82466b3c..db039f9e 100644 --- a/src/ocrmypdf/fpdf_renderer/__init__.py +++ b/src/ocrmypdf/fpdf_renderer/__init__.py @@ -6,6 +6,7 @@ This module provides the PDF renderer using fpdf2 for creating searchable OCR text layers. """ +from __future__ import annotations from ocrmypdf.fpdf_renderer.renderer import ( DebugRenderOptions, diff --git a/src/ocrmypdf/helpers.py b/src/ocrmypdf/helpers.py index 7b810de2..9884e245 100644 --- a/src/ocrmypdf/helpers.py +++ b/src/ocrmypdf/helpers.py @@ -25,7 +25,6 @@ from typing import ( import img2pdf import pikepdf -from deprecation import deprecated log = logging.getLogger(__name__) diff --git a/src/ocrmypdf/hocrtransform/__main__.py b/src/ocrmypdf/hocrtransform/__main__.py index df561174..76375293 100644 --- a/src/ocrmypdf/hocrtransform/__main__.py +++ b/src/ocrmypdf/hocrtransform/__main__.py @@ -2,6 +2,7 @@ # SPDX-License-Identifier: MIT """Simple CLI for testing HOCR to PDF conversion using fpdf2 renderer.""" +from __future__ import annotations import argparse from pathlib import Path diff --git a/src/ocrmypdf/languages.py b/src/ocrmypdf/languages.py index 90e36775..0702fe53 100644 --- a/src/ocrmypdf/languages.py +++ b/src/ocrmypdf/languages.py @@ -6,6 +6,7 @@ Derived from https://www.loc.gov/standards/iso639-2/ascii_8bits.html """ +from __future__ import annotations from typing import NamedTuple diff --git a/tests/conftest.py b/tests/conftest.py index 05088071..472aaee0 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -23,9 +23,10 @@ class Gs106WarningFilter(logging.Filter): def filter(self, record: logging.LogRecord) -> bool: # Allow all records except the expected Ghostscript 10.6.x warning - if "Ghostscript 10.6.x contains JPEG encoding errors" in record.getMessage(): - return False - return True + return ( + "Ghostscript 10.6.x contains JPEG encoding errors" + not in record.getMessage() + ) @pytest.fixture(autouse=True) diff --git a/tests/test_json_serialization.py b/tests/test_json_serialization.py index 3de76ed5..e08a8bfe 100644 --- a/tests/test_json_serialization.py +++ b/tests/test_json_serialization.py @@ -1,4 +1,5 @@ """Test JSON serialization of OcrOptions for multiprocessing compatibility.""" +from __future__ import annotations import multiprocessing from io import BytesIO diff --git a/tests/test_metadata.py b/tests/test_metadata.py index 80da5f6b..be8826bb 100644 --- a/tests/test_metadata.py +++ b/tests/test_metadata.py @@ -3,9 +3,8 @@ from __future__ import annotations -import datetime +import datetime as dt import warnings -from datetime import timezone from shutil import copyfile import pikepdf @@ -198,10 +197,7 @@ def test_creation_date_preserved(output_type, resources, infile, outpdf): # We expect that the modified date is quite recent date_after = decode_pdf_date(str(after['/ModDate'])) - assert ( - seconds_between_dates(date_after, datetime.datetime.now(timezone.utc)) - < 1000 - ) + assert seconds_between_dates(date_after, dt.datetime.now(dt.UTC)) < 1000 @pytest.fixture diff --git a/tests/test_multilingual_direct.py b/tests/test_multilingual_direct.py index 0a5fe35d..113e29d2 100644 --- a/tests/test_multilingual_direct.py +++ b/tests/test_multilingual_direct.py @@ -10,6 +10,7 @@ This tests the fpdf2 renderer with various language groups: - CJK (Chinese Simplified/Traditional, Japanese, Korean) - Devanagari (Hindi, Sanskrit) """ +from __future__ import annotations import shutil import subprocess diff --git a/tests/test_unpaper.py b/tests/test_unpaper.py index 31425cba..319748da 100644 --- a/tests/test_unpaper.py +++ b/tests/test_unpaper.py @@ -119,7 +119,7 @@ def test_unpaper_args_invalid(resources, outpdf): def test_unpaper_image_too_big(resources, outdir, caplog): with patch('ocrmypdf._exec.unpaper.UNPAPER_IMAGE_PIXEL_LIMIT', 42): infile = resources / 'crom.png' - unpaper.clean(infile, outdir / 'out.png', dpi=300) == infile + assert unpaper.clean(infile, outdir / 'out.png', dpi=300) == infile assert any( 'too large for cleaning' in rec.message diff --git a/tests/test_watcher.py b/tests/test_watcher.py index 40b01089..8cfcaf7e 100644 --- a/tests/test_watcher.py +++ b/tests/test_watcher.py @@ -1,4 +1,6 @@ -import datetime +from __future__ import annotations + +import datetime as dt import os import shutil import subprocess @@ -43,8 +45,8 @@ def test_watcher(tmp_path, resources, year_month): if year_month: assert ( output_dir - / f'{datetime.date.today().year}' - / f'{datetime.date.today().month:02d}' + / f'{dt.date.today().year}' + / f'{dt.date.today().month:02d}' / 'trivial.pdf' ).exists() else: